{
 "note": "Receipts copied from public-runs/provider-h2h-hard. Every attempt is kept, failures included.",
 "cases": [
  {
   "id": "merge-ranges",
   "title": "Fix an interval-merge function (off-by-one and edge cases)",
   "kind": "code-fix",
   "difficulty": "hard",
   "format": "js",
   "validator": "sandbox"
  },
  {
   "id": "day-hours",
   "title": "Fix a time-zone day-length function (DST)",
   "kind": "code-fix",
   "difficulty": "hard",
   "format": "js",
   "validator": "sandbox"
  },
  {
   "id": "csv-parse",
   "title": "Write a CSV parser (quoted newlines, strict errors)",
   "kind": "code-write",
   "difficulty": "hard",
   "format": "js",
   "validator": "sandbox"
  },
  {
   "id": "event-loop-order",
   "title": "Predict JavaScript event-loop output order",
   "kind": "reasoning",
   "difficulty": "hard",
   "format": "line",
   "validator": "exact",
   "expected": "A1,S,M1,A2,Q1,TH,M3,M2,Q2,B1,M4,R1,T1,T2,T2M,A3"
  },
  {
   "id": "talk-schedule",
   "title": "Solve a multi-constraint room schedule",
   "kind": "constraint",
   "difficulty": "hard",
   "format": "json",
   "validator": "sandbox"
  },
  {
   "id": "semver-regex",
   "title": "Write a strict SemVer 2.0.0 regex",
   "kind": "regex",
   "difficulty": "hard",
   "format": "regex",
   "validator": "sandbox"
  },
  {
   "id": "money-refactor",
   "title": "Refactor to remove duplication, keep 20 tests green",
   "kind": "refactor",
   "difficulty": "hard",
   "format": "js",
   "validator": "sandbox"
  },
  {
   "id": "q1-sql",
   "title": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "kind": "sql",
   "difficulty": "hard",
   "format": "sql",
   "validator": "sandbox"
  }
 ],
 "controls": {
  "ranAt": "2026-10-06T03:23:22.492Z",
  "allOk": true,
  "cases": [
   {
    "caseId": "merge-ranges",
    "referencePassed": true,
    "referenceChecks": 18,
    "wrongAnswers": 5,
    "wrongAnswersFailed": 5,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "day-hours",
    "referencePassed": true,
    "referenceChecks": 36,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "csv-parse",
    "referencePassed": true,
    "referenceChecks": 22,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "event-loop-order",
    "referencePassed": true,
    "referenceChecks": 1,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "talk-schedule",
    "referencePassed": true,
    "referenceChecks": 14,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "semver-regex",
    "referencePassed": true,
    "referenceChecks": 30,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "money-refactor",
    "referencePassed": true,
    "referenceChecks": 25,
    "wrongAnswers": 2,
    "wrongAnswersFailed": 2,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "q1-sql",
    "referencePassed": true,
    "referenceChecks": 8,
    "wrongAnswers": 4,
    "wrongAnswersFailed": 4,
    "wrappedReferenceIsFormatMiss": true
   }
  ]
 },
 "batches": [
  {
   "batch": "batch-1",
   "schema": "agent-provider-h2h@2",
   "route": "claude-cli",
   "startedAt": "2026-10-06T03:23:59.224Z",
   "finishedAt": "2026-10-06T04:02:39.469Z",
   "stopReason": null,
   "trimmed": [],
   "generatedAt": null
  },
  {
   "batch": "batch-2",
   "schema": "agent-provider-h2h@2",
   "route": "codex-cli",
   "startedAt": "2026-10-06T03:23:59.235Z",
   "finishedAt": "2026-10-06T03:24:09.934Z",
   "stopReason": null,
   "trimmed": [],
   "generatedAt": null
  },
  {
   "batch": "batch-3",
   "schema": "agent-provider-h2h@2",
   "route": "codex-cli",
   "startedAt": "2026-10-06T12:35:37.505Z",
   "finishedAt": "2026-10-06T13:15:04.544Z",
   "stopReason": null,
   "trimmed": [],
   "generatedAt": null
  }
 ],
 "receipts": [
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:23:59.224Z",
   "finishedAt": "2026-10-06T03:24:18.255Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 17140,
   "totalMs": 18527,
   "modelTurnMs": 17874,
   "inputTokens": 3906,
   "outputTokens": 2169,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 1977,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 440,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:18.256Z",
   "finishedAt": "2026-10-06T03:24:21.682Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1563,
   "totalMs": 2932,
   "modelTurnMs": 1913,
   "inputTokens": 2236,
   "outputTokens": 220,
   "cacheReadTokens": 531,
   "cacheWriteTokens": 1703,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 451,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:21.682Z",
   "finishedAt": "2026-10-06T03:24:26.403Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2394,
   "totalMs": 4235,
   "modelTurnMs": 3457,
   "inputTokens": 2232,
   "outputTokens": 323,
   "cacheReadTokens": 531,
   "cacheWriteTokens": 1699,
   "reasoningTokens": 100,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 433,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:26.403Z",
   "finishedAt": "2026-10-06T03:24:31.649Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2372,
   "totalMs": 4762,
   "modelTurnMs": 4004,
   "inputTokens": 4090,
   "outputTokens": 318,
   "cacheReadTokens": 531,
   "cacheWriteTokens": 3557,
   "reasoningTokens": 85,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 457,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:31.649Z",
   "finishedAt": "2026-10-06T03:24:37.494Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3298,
   "totalMs": 5357,
   "modelTurnMs": 4437,
   "inputTokens": 2230,
   "outputTokens": 422,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 765,
   "reasoningTokens": 188,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 456,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:37.494Z",
   "finishedAt": "2026-10-06T03:25:17.017Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 34370,
   "totalMs": 39004,
   "modelTurnMs": 38386,
   "inputTokens": 3932,
   "outputTokens": 5074,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 4372,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 36,
   "failures": [
    "2026-03-08 America/New_York (host Asia/Tokyo): got 22.999969482421875, want 23",
    "2026-11-01 America/New_York (host Asia/Tokyo): got 25.000030517578125, want 25",
    "2026-03-29 Europe/London (host Asia/Tokyo): got 23.00006103515625, want 23",
    "2026-10-25 Europe/London (host Asia/Tokyo): got 24.99993896484375, want 25",
    "2026-04-05 Australia/Lord_Howe (host Asia/Tokyo): got 24.499969482421875, want 24.5"
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 36,
     "failures": [
      "2026-03-08 America/New_York (host Asia/Tokyo): got 22.999969482421875, want 23",
      "2026-11-01 America/New_York (host Asia/Tokyo): got 25.000030517578125, want 25",
      "2026-03-29 Europe/London (host Asia/Tokyo): got 23.00006103515625, want 23",
      "2026-10-25 Europe/London (host Asia/Tokyo): got 24.99993896484375, want 25",
      "2026-04-05 Australia/Lord_Howe (host Asia/Tokyo): got 24.499969482421875, want 24.5"
     ],
     "failureCount": 12
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1900,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failureCount": 30,
   "failedOutputLastLine": "}",
   "failedOutputLineCount": 53
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:25:17.017Z",
   "finishedAt": "2026-10-06T03:25:39.099Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 18132,
   "totalMs": 21610,
   "modelTurnMs": 20990,
   "inputTokens": 2260,
   "outputTokens": 1964,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 795,
   "reasoningTokens": 1249,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1504,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:25:39.099Z",
   "finishedAt": "2026-10-06T03:26:05.893Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 20394,
   "totalMs": 26304,
   "modelTurnMs": 25712,
   "inputTokens": 2256,
   "outputTokens": 2444,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 791,
   "reasoningTokens": 1580,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1888,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:26:05.894Z",
   "finishedAt": "2026-10-06T03:26:40.078Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 27424,
   "totalMs": 33714,
   "modelTurnMs": 33092,
   "inputTokens": 4115,
   "outputTokens": 2395,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 791,
   "reasoningTokens": 1815,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1182,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:26:40.079Z",
   "finishedAt": "2026-10-06T03:27:43.559Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 56228,
   "totalMs": 63002,
   "modelTurnMs": 62404,
   "inputTokens": 2256,
   "outputTokens": 4052,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 791,
   "reasoningTokens": 3301,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1586,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:27:43.560Z",
   "finishedAt": "2026-10-06T03:28:17.791Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 29753,
   "totalMs": 33807,
   "modelTurnMs": 33212,
   "inputTokens": 3928,
   "outputTokens": 4211,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 3590,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1881,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:28:17.791Z",
   "finishedAt": "2026-10-06T03:28:27.073Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6100,
   "totalMs": 8845,
   "modelTurnMs": 8193,
   "inputTokens": 2272,
   "outputTokens": 1112,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 807,
   "reasoningTokens": 551,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1432,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:28:27.074Z",
   "finishedAt": "2026-10-06T03:28:38.496Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7161,
   "totalMs": 10978,
   "modelTurnMs": 10149,
   "inputTokens": 2268,
   "outputTokens": 1071,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 803,
   "reasoningTokens": 543,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1373,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:28:38.496Z",
   "finishedAt": "2026-10-06T03:28:55.149Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11267,
   "totalMs": 16197,
   "modelTurnMs": 15556,
   "inputTokens": 4126,
   "outputTokens": 1452,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 802,
   "reasoningTokens": 903,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1409,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:28:55.149Z",
   "finishedAt": "2026-10-06T03:29:06.709Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6917,
   "totalMs": 11102,
   "modelTurnMs": 10271,
   "inputTokens": 2268,
   "outputTokens": 1108,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 803,
   "reasoningTokens": 528,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1514,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:29:06.709Z",
   "finishedAt": "2026-10-06T03:29:32.870Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 25616,
   "totalMs": 25963,
   "modelTurnMs": 25322,
   "inputTokens": 3951,
   "outputTokens": 3831,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 3781,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 44,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "A1,S,M1,A2,Q1,TH,M2,M3,M4,Q2,B1,R1,T1,T2,T2M",
   "failedOutputLineCount": 1
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:29:32.871Z",
   "finishedAt": "2026-10-06T03:29:41.258Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7734,
   "totalMs": 8168,
   "modelTurnMs": 7527,
   "inputTokens": 2290,
   "outputTokens": 1115,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 825,
   "reasoningTokens": 1064,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:29:41.258Z",
   "finishedAt": "2026-10-06T03:29:53.372Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11673,
   "totalMs": 11910,
   "modelTurnMs": 11242,
   "inputTokens": 2286,
   "outputTokens": 1009,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 821,
   "reasoningTokens": 958,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:29:53.373Z",
   "finishedAt": "2026-10-06T03:30:17.484Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 23183,
   "totalMs": 23899,
   "modelTurnMs": 23191,
   "inputTokens": 4145,
   "outputTokens": 1816,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 821,
   "reasoningTokens": 1765,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:30:17.484Z",
   "finishedAt": "2026-10-06T03:30:31.280Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13001,
   "totalMs": 13598,
   "modelTurnMs": 12930,
   "inputTokens": 2286,
   "outputTokens": 1325,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 821,
   "reasoningTokens": 1274,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:30:31.280Z",
   "finishedAt": "2026-10-06T03:31:27.120Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 53767,
   "totalMs": 54643,
   "modelTurnMs": 54012,
   "inputTokens": 4033,
   "outputTokens": 6350,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 6236,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 14,
   "failures": [
    "json: output is not valid JSON"
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 14,
     "failures": [
      "json: output is not valid JSON"
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 229,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "```",
   "failedOutputLineCount": 3
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:31:27.121Z",
   "finishedAt": "2026-10-06T03:31:35.082Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6683,
   "totalMs": 7362,
   "modelTurnMs": 6608,
   "inputTokens": 2378,
   "outputTokens": 848,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 913,
   "reasoningTokens": 716,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:31:35.083Z",
   "finishedAt": "2026-10-06T03:31:43.489Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6711,
   "totalMs": 7815,
   "modelTurnMs": 7067,
   "inputTokens": 2376,
   "outputTokens": 706,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 911,
   "reasoningTokens": 574,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:31:43.489Z",
   "finishedAt": "2026-10-06T03:32:01.363Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 16108,
   "totalMs": 17271,
   "modelTurnMs": 16613,
   "inputTokens": 4234,
   "outputTokens": 1222,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 910,
   "reasoningTokens": 1090,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:32:01.363Z",
   "finishedAt": "2026-10-06T03:32:10.858Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7774,
   "totalMs": 8900,
   "modelTurnMs": 8107,
   "inputTokens": 2375,
   "outputTokens": 788,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 910,
   "reasoningTokens": 656,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:32:10.859Z",
   "finishedAt": "2026-10-06T03:33:15.829Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 63301,
   "totalMs": 64478,
   "modelTurnMs": 63754,
   "inputTokens": 3879,
   "outputTokens": 7659,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 7498,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 197,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:33:15.830Z",
   "finishedAt": "2026-10-06T03:33:18.549Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 906,
   "totalMs": 2259,
   "modelTurnMs": 1436,
   "inputTokens": 2236,
   "outputTokens": 176,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 771,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:33:18.549Z",
   "finishedAt": "2026-10-06T03:33:24.388Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3699,
   "totalMs": 5348,
   "modelTurnMs": 4407,
   "inputTokens": 2232,
   "outputTokens": 482,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 767,
   "reasoningTokens": 300,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 181,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:33:24.388Z",
   "finishedAt": "2026-10-06T03:33:34.454Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7419,
   "totalMs": 9601,
   "modelTurnMs": 8870,
   "inputTokens": 4091,
   "outputTokens": 443,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 767,
   "reasoningTokens": 261,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 183,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:33:34.454Z",
   "finishedAt": "2026-10-06T03:33:39.948Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3458,
   "totalMs": 4999,
   "modelTurnMs": 4221,
   "inputTokens": 2233,
   "outputTokens": 475,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 768,
   "reasoningTokens": 293,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 181,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:33:39.949Z",
   "finishedAt": "2026-10-06T03:33:56.371Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12884,
   "totalMs": 15906,
   "modelTurnMs": 15211,
   "inputTokens": 4221,
   "outputTokens": 1899,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 4211,
   "reasoningTokens": 1452,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1287,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:33:56.372Z",
   "finishedAt": "2026-10-06T03:34:00.451Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1338,
   "totalMs": 3573,
   "modelTurnMs": 2747,
   "inputTokens": 2667,
   "outputTokens": 429,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1202,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 923,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:34:00.451Z",
   "finishedAt": "2026-10-06T03:34:09.127Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5002,
   "totalMs": 8153,
   "modelTurnMs": 7431,
   "inputTokens": 2664,
   "outputTokens": 880,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1199,
   "reasoningTokens": 429,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1007,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:34:09.127Z",
   "finishedAt": "2026-10-06T03:34:22.614Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8675,
   "totalMs": 12912,
   "modelTurnMs": 12214,
   "inputTokens": 4524,
   "outputTokens": 1135,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 1200,
   "reasoningTokens": 646,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1085,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:34:22.615Z",
   "finishedAt": "2026-10-06T03:34:32.426Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5914,
   "totalMs": 9117,
   "modelTurnMs": 8375,
   "inputTokens": 2664,
   "outputTokens": 995,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1199,
   "reasoningTokens": 528,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1044,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:34:32.427Z",
   "finishedAt": "2026-10-06T03:35:12.361Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 34636,
   "totalMs": 38890,
   "modelTurnMs": 38155,
   "inputTokens": 4023,
   "outputTokens": 4969,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 4305,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 1,
   "failures": [
    "format: output does not start with SELECT or WITH"
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "format: output does not start with SELECT or WITH"
     ]
    },
    "lenient": {
     "extractor": "fenced-block",
     "passed": true,
     "checks": 8,
     "failures": []
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1654,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "```",
   "failedOutputLineCount": 58
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:35:12.362Z",
   "finishedAt": "2026-10-06T03:35:21.956Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5800,
   "totalMs": 8915,
   "modelTurnMs": 8116,
   "inputTokens": 2518,
   "outputTokens": 1470,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1053,
   "reasoningTokens": 799,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1254,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:35:21.957Z",
   "finishedAt": "2026-10-06T03:35:35.338Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7545,
   "totalMs": 12779,
   "modelTurnMs": 12010,
   "inputTokens": 2513,
   "outputTokens": 1552,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1048,
   "reasoningTokens": 796,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1368,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T03:35:35.339Z",
   "finishedAt": "2026-10-06T03:35:52.005Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9701,
   "totalMs": 16066,
   "modelTurnMs": 15373,
   "inputTokens": 4372,
   "outputTokens": 1626,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 1048,
   "reasoningTokens": 875,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1378,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:35:52.006Z",
   "finishedAt": "2026-10-06T03:36:06.285Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8420,
   "totalMs": 13663,
   "modelTurnMs": 12823,
   "inputTokens": 2515,
   "outputTokens": 1694,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1050,
   "reasoningTokens": 920,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1396,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:36:06.286Z",
   "finishedAt": "2026-10-06T03:36:31.610Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 23401,
   "totalMs": 24709,
   "modelTurnMs": 24053,
   "inputTokens": 3907,
   "outputTokens": 2797,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 2606,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 430,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:36:31.610Z",
   "finishedAt": "2026-10-06T03:36:35.301Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1328,
   "totalMs": 3064,
   "modelTurnMs": 2134,
   "inputTokens": 2235,
   "outputTokens": 219,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 770,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 449,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:36:35.302Z",
   "finishedAt": "2026-10-06T03:36:40.564Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2826,
   "totalMs": 4646,
   "modelTurnMs": 3735,
   "inputTokens": 2231,
   "outputTokens": 336,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 766,
   "reasoningTokens": 114,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 431,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:36:40.564Z",
   "finishedAt": "2026-10-06T03:36:45.635Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2002,
   "totalMs": 4459,
   "modelTurnMs": 3600,
   "inputTokens": 4090,
   "outputTokens": 320,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 766,
   "reasoningTokens": 75,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 479,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:36:45.636Z",
   "finishedAt": "2026-10-06T03:36:51.751Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3986,
   "totalMs": 5520,
   "modelTurnMs": 4600,
   "inputTokens": 2231,
   "outputTokens": 404,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 766,
   "reasoningTokens": 182,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 431,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:36:51.752Z",
   "finishedAt": "2026-10-06T03:37:19.342Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 23670,
   "totalMs": 26154,
   "modelTurnMs": 25465,
   "inputTokens": 3931,
   "outputTokens": 4071,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 3515,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 0,
   "failures": [
    "load: Unexpected identifier '$'"
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 0,
     "failures": [
      "load: Unexpected identifier '$'"
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1401,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "```",
   "failedOutputLineCount": 45
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:37:19.343Z",
   "finishedAt": "2026-10-06T03:37:39.635Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 16748,
   "totalMs": 19618,
   "modelTurnMs": 18846,
   "inputTokens": 2259,
   "outputTokens": 2185,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 794,
   "reasoningTokens": 1614,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1221,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:37:39.636Z",
   "finishedAt": "2026-10-06T03:38:07.517Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 21771,
   "totalMs": 27212,
   "modelTurnMs": 26522,
   "inputTokens": 2256,
   "outputTokens": 2531,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 791,
   "reasoningTokens": 1778,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1666,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:38:07.518Z",
   "finishedAt": "2026-10-06T03:39:38.057Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 85331,
   "totalMs": 89997,
   "modelTurnMs": 89189,
   "inputTokens": 4114,
   "outputTokens": 6465,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 790,
   "reasoningTokens": 5889,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1259,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:39:38.057Z",
   "finishedAt": "2026-10-06T03:40:14.987Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 30640,
   "totalMs": 36440,
   "modelTurnMs": 35789,
   "inputTokens": 2255,
   "outputTokens": 3558,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 790,
   "reasoningTokens": 2756,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1648,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:40:14.988Z",
   "finishedAt": "2026-10-06T03:40:54.460Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 35423,
   "totalMs": 39012,
   "modelTurnMs": 38220,
   "inputTokens": 3926,
   "outputTokens": 5053,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 4497,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1696,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:40:54.461Z",
   "finishedAt": "2026-10-06T03:41:04.504Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6918,
   "totalMs": 9557,
   "modelTurnMs": 8751,
   "inputTokens": 2272,
   "outputTokens": 1260,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 807,
   "reasoningTokens": 722,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1440,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:41:04.504Z",
   "finishedAt": "2026-10-06T03:41:16.443Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7152,
   "totalMs": 11470,
   "modelTurnMs": 10737,
   "inputTokens": 2268,
   "outputTokens": 1154,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 803,
   "reasoningTokens": 524,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1756,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:41:16.443Z",
   "finishedAt": "2026-10-06T03:41:38.256Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 16310,
   "totalMs": 21381,
   "modelTurnMs": 20784,
   "inputTokens": 4127,
   "outputTokens": 1632,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 803,
   "reasoningTokens": 1088,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1463,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:41:38.257Z",
   "finishedAt": "2026-10-06T03:41:50.899Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6682,
   "totalMs": 12217,
   "modelTurnMs": 11529,
   "inputTokens": 2267,
   "outputTokens": 1366,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 802,
   "reasoningTokens": 573,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 2019,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:41:50.899Z",
   "finishedAt": "2026-10-06T03:42:45.110Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 53626,
   "totalMs": 54021,
   "modelTurnMs": 53454,
   "inputTokens": 3949,
   "outputTokens": 6627,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 6575,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "A1,TH,S,M1,A2,Q1,B1,M3,M4,R1,M2,Q2,T1,T2,T2M,A3",
   "failedOutputLineCount": 1
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:42:45.111Z",
   "finishedAt": "2026-10-06T03:42:55.184Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9375,
   "totalMs": 9865,
   "modelTurnMs": 9173,
   "inputTokens": 2290,
   "outputTokens": 1248,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 825,
   "reasoningTokens": 1197,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:42:55.185Z",
   "finishedAt": "2026-10-06T03:43:05.576Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9675,
   "totalMs": 10198,
   "modelTurnMs": 9581,
   "inputTokens": 2285,
   "outputTokens": 1066,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 820,
   "reasoningTokens": 1015,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:43:05.577Z",
   "finishedAt": "2026-10-06T03:43:22.558Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 16218,
   "totalMs": 16779,
   "modelTurnMs": 16166,
   "inputTokens": 4145,
   "outputTokens": 1279,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 821,
   "reasoningTokens": 1228,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:43:22.559Z",
   "finishedAt": "2026-10-06T03:43:36.215Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12930,
   "totalMs": 13454,
   "modelTurnMs": 12789,
   "inputTokens": 2287,
   "outputTokens": 1351,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 822,
   "reasoningTokens": 1300,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:43:36.216Z",
   "finishedAt": "2026-10-06T03:44:45.681Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 67878,
   "totalMs": 68772,
   "modelTurnMs": 68167,
   "inputTokens": 4032,
   "outputTokens": 8069,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 7954,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 14,
   "failures": [
    "json: output is not valid JSON"
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 14,
     "failures": [
      "json: output is not valid JSON"
     ]
    },
    "lenient": {
     "extractor": "fenced-block",
     "passed": true,
     "checks": 14,
     "failures": []
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 229,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "```",
   "failedOutputLineCount": 3
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:44:45.682Z",
   "finishedAt": "2026-10-06T03:44:53.883Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6891,
   "totalMs": 7762,
   "modelTurnMs": 7062,
   "inputTokens": 2379,
   "outputTokens": 868,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 914,
   "reasoningTokens": 736,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:44:53.884Z",
   "finishedAt": "2026-10-06T03:45:01.943Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6561,
   "totalMs": 7627,
   "modelTurnMs": 7014,
   "inputTokens": 2374,
   "outputTokens": 665,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 909,
   "reasoningTokens": 533,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:45:01.943Z",
   "finishedAt": "2026-10-06T03:45:13.606Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9907,
   "totalMs": 11233,
   "modelTurnMs": 10561,
   "inputTokens": 4234,
   "outputTokens": 918,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 910,
   "reasoningTokens": 786,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:45:13.606Z",
   "finishedAt": "2026-10-06T03:45:22.599Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7480,
   "totalMs": 8541,
   "modelTurnMs": 7837,
   "inputTokens": 2375,
   "outputTokens": 787,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 910,
   "reasoningTokens": 655,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:45:22.600Z",
   "finishedAt": "2026-10-06T03:46:17.532Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 53381,
   "totalMs": 54492,
   "modelTurnMs": 53875,
   "inputTokens": 3879,
   "outputTokens": 6692,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 6538,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 189,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:46:17.533Z",
   "finishedAt": "2026-10-06T03:46:21.623Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2479,
   "totalMs": 3668,
   "modelTurnMs": 2882,
   "inputTokens": 2235,
   "outputTokens": 449,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 770,
   "reasoningTokens": 267,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:46:21.624Z",
   "finishedAt": "2026-10-06T03:46:26.795Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3238,
   "totalMs": 4747,
   "modelTurnMs": 3918,
   "inputTokens": 2232,
   "outputTokens": 426,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 767,
   "reasoningTokens": 244,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 181,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:46:26.796Z",
   "finishedAt": "2026-10-06T03:46:34.740Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4860,
   "totalMs": 7517,
   "modelTurnMs": 6856,
   "inputTokens": 4091,
   "outputTokens": 412,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 767,
   "reasoningTokens": 236,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:46:34.741Z",
   "finishedAt": "2026-10-06T03:46:38.843Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2152,
   "totalMs": 3625,
   "modelTurnMs": 2881,
   "inputTokens": 2232,
   "outputTokens": 285,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 767,
   "reasoningTokens": 103,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 181,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:46:38.844Z",
   "finishedAt": "2026-10-06T03:46:54.575Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13216,
   "totalMs": 15265,
   "modelTurnMs": 14570,
   "inputTokens": 4221,
   "outputTokens": 2391,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 4211,
   "reasoningTokens": 1955,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 25,
   "failures": [
    "structure:one-padStart-cents: found 2, want at most 1"
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 25,
     "failures": [
      "structure:one-padStart-cents: found 2, want at most 1"
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1247,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "}",
   "failedOutputLineCount": 33
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:46:54.576Z",
   "finishedAt": "2026-10-06T03:47:02.242Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4983,
   "totalMs": 7182,
   "modelTurnMs": 6489,
   "inputTokens": 2669,
   "outputTokens": 996,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1204,
   "reasoningTokens": 545,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1005,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:47:02.243Z",
   "finishedAt": "2026-10-06T03:47:09.930Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4592,
   "totalMs": 7202,
   "modelTurnMs": 6507,
   "inputTokens": 2664,
   "outputTokens": 688,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1199,
   "reasoningTokens": 214,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1049,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:47:09.930Z",
   "finishedAt": "2026-10-06T03:47:24.086Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9619,
   "totalMs": 13675,
   "modelTurnMs": 13020,
   "inputTokens": 4523,
   "outputTokens": 1238,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 1199,
   "reasoningTokens": 796,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 999,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:47:24.087Z",
   "finishedAt": "2026-10-06T03:47:33.367Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5492,
   "totalMs": 8841,
   "modelTurnMs": 7600,
   "inputTokens": 2664,
   "outputTokens": 897,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1199,
   "reasoningTokens": 431,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1026,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:47:33.368Z",
   "finishedAt": "2026-10-06T03:48:30.295Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 52094,
   "totalMs": 56130,
   "modelTurnMs": 55510,
   "inputTokens": 4023,
   "outputTokens": 7319,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 6691,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 1,
   "failures": [
    "format: output does not start with SELECT or WITH"
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "format: output does not start with SELECT or WITH"
     ]
    },
    "lenient": {
     "extractor": "fenced-block",
     "passed": true,
     "checks": 8,
     "failures": []
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1581,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "```",
   "failedOutputLineCount": 62
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:48:30.295Z",
   "finishedAt": "2026-10-06T03:48:39.634Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5333,
   "totalMs": 8467,
   "modelTurnMs": 7675,
   "inputTokens": 2518,
   "outputTokens": 1268,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1053,
   "reasoningTokens": 619,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1195,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:48:39.635Z",
   "finishedAt": "2026-10-06T03:48:52.599Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7487,
   "totalMs": 12440,
   "modelTurnMs": 11705,
   "inputTokens": 2514,
   "outputTokens": 1519,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1049,
   "reasoningTokens": 781,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1309,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T03:48:52.600Z",
   "finishedAt": "2026-10-06T03:49:13.701Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11995,
   "totalMs": 20307,
   "modelTurnMs": 19532,
   "inputTokens": 4371,
   "outputTokens": 1675,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 1047,
   "reasoningTokens": 865,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1528,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:49:13.701Z",
   "finishedAt": "2026-10-06T03:49:26.661Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7334,
   "totalMs": 12456,
   "modelTurnMs": 11745,
   "inputTokens": 2514,
   "outputTokens": 1509,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1049,
   "reasoningTokens": 739,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1402,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:49:26.663Z",
   "finishedAt": "2026-10-06T03:49:48.389Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 20306,
   "totalMs": 21218,
   "modelTurnMs": 20550,
   "inputTokens": 3907,
   "outputTokens": 2874,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 2703,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 393,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:49:48.390Z",
   "finishedAt": "2026-10-06T03:49:51.212Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 859,
   "totalMs": 2380,
   "modelTurnMs": 1566,
   "inputTokens": 2234,
   "outputTokens": 219,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 769,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 449,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:49:51.213Z",
   "finishedAt": "2026-10-06T03:49:56.107Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2558,
   "totalMs": 4445,
   "modelTurnMs": 3587,
   "inputTokens": 2231,
   "outputTokens": 344,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 766,
   "reasoningTokens": 122,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 431,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:49:56.108Z",
   "finishedAt": "2026-10-06T03:50:11.814Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12401,
   "totalMs": 15205,
   "modelTurnMs": 14496,
   "inputTokens": 4090,
   "outputTokens": 449,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 766,
   "reasoningTokens": 197,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 495,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T03:50:11.815Z",
   "finishedAt": "2026-10-06T03:50:18.059Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3722,
   "totalMs": 5733,
   "modelTurnMs": 4913,
   "inputTokens": 2231,
   "outputTokens": 466,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 766,
   "reasoningTokens": 221,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 482,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:50:18.060Z",
   "finishedAt": "2026-10-06T03:50:59.154Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 35654,
   "totalMs": 40564,
   "modelTurnMs": 39897,
   "inputTokens": 3930,
   "outputTokens": 5367,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 4614,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1825,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:50:59.155Z",
   "finishedAt": "2026-10-06T03:51:34.467Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 30571,
   "totalMs": 34788,
   "modelTurnMs": 34051,
   "inputTokens": 2259,
   "outputTokens": 3895,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 794,
   "reasoningTokens": 3060,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1777,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:51:34.468Z",
   "finishedAt": "2026-10-06T03:51:54.728Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 14537,
   "totalMs": 19739,
   "modelTurnMs": 19083,
   "inputTokens": 2256,
   "outputTokens": 1890,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 791,
   "reasoningTokens": 1136,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1594,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:51:54.729Z",
   "finishedAt": "2026-10-06T03:52:20.323Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 19710,
   "totalMs": 25093,
   "modelTurnMs": 23991,
   "inputTokens": 4117,
   "outputTokens": 2007,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 793,
   "reasoningTokens": 1427,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1155,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T03:52:20.324Z",
   "finishedAt": "2026-10-06T03:52:59.616Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 34485,
   "totalMs": 38793,
   "modelTurnMs": 38172,
   "inputTokens": 2256,
   "outputTokens": 3659,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 791,
   "reasoningTokens": 3027,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1286,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:52:59.618Z",
   "finishedAt": "2026-10-06T03:54:15.559Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 70309,
   "totalMs": 75129,
   "modelTurnMs": 74409,
   "inputTokens": 3927,
   "outputTokens": 9321,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 8569,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 0,
   "failures": [
    "load: \"\" is not a function"
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 0,
     "failures": [
      "load: \"\" is not a function"
     ]
    },
    "lenient": {
     "extractor": "fenced-block",
     "passed": true,
     "checks": 22,
     "failures": []
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 2608,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "```",
   "failedOutputLineCount": 98
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:54:15.560Z",
   "finishedAt": "2026-10-06T03:54:26.664Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7894,
   "totalMs": 10611,
   "modelTurnMs": 9899,
   "inputTokens": 2271,
   "outputTokens": 1104,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 806,
   "reasoningTokens": 547,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1499,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:54:26.665Z",
   "finishedAt": "2026-10-06T03:54:38.783Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6854,
   "totalMs": 11666,
   "modelTurnMs": 10923,
   "inputTokens": 2270,
   "outputTokens": 1131,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 805,
   "reasoningTokens": 440,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1840,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:54:38.783Z",
   "finishedAt": "2026-10-06T03:54:55.120Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11198,
   "totalMs": 15878,
   "modelTurnMs": 15273,
   "inputTokens": 4128,
   "outputTokens": 1539,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 804,
   "reasoningTokens": 1000,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1392,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T03:54:55.121Z",
   "finishedAt": "2026-10-06T03:55:08.184Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9063,
   "totalMs": 12615,
   "modelTurnMs": 11898,
   "inputTokens": 2268,
   "outputTokens": 1310,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 803,
   "reasoningTokens": 800,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1286,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:55:08.185Z",
   "finishedAt": "2026-10-06T03:56:05.028Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 56167,
   "totalMs": 56645,
   "modelTurnMs": 56024,
   "inputTokens": 3950,
   "outputTokens": 7117,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 7065,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "S,A1,TH,M1,A2,Q1,B1,M3,M4,R1,M2,Q2,T1,A3,T2,T2M",
   "failedOutputLineCount": 1
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:56:05.029Z",
   "finishedAt": "2026-10-06T03:56:14.959Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9145,
   "totalMs": 9573,
   "modelTurnMs": 8790,
   "inputTokens": 2290,
   "outputTokens": 1228,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 825,
   "reasoningTokens": 1177,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:56:14.960Z",
   "finishedAt": "2026-10-06T03:56:26.796Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10928,
   "totalMs": 11480,
   "modelTurnMs": 10615,
   "inputTokens": 2286,
   "outputTokens": 1160,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 821,
   "reasoningTokens": 1109,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:56:26.797Z",
   "finishedAt": "2026-10-06T03:56:53.954Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 26726,
   "totalMs": 26908,
   "modelTurnMs": 26216,
   "inputTokens": 4145,
   "outputTokens": 1657,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 821,
   "reasoningTokens": 1606,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T03:56:53.955Z",
   "finishedAt": "2026-10-06T03:57:07.663Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12917,
   "totalMs": 13503,
   "modelTurnMs": 12767,
   "inputTokens": 2285,
   "outputTokens": 1249,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 820,
   "reasoningTokens": 1198,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:57:07.664Z",
   "finishedAt": "2026-10-06T03:57:46.383Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 37303,
   "totalMs": 37968,
   "modelTurnMs": 37297,
   "inputTokens": 4033,
   "outputTokens": 5037,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 4922,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 14,
   "failures": [
    "json: output is not valid JSON"
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 14,
     "failures": [
      "json: output is not valid JSON"
     ]
    },
    "lenient": {
     "extractor": "fenced-block",
     "passed": true,
     "checks": 14,
     "failures": []
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 229,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "```",
   "failedOutputLineCount": 3
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:57:46.384Z",
   "finishedAt": "2026-10-06T03:57:54.596Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7003,
   "totalMs": 7737,
   "modelTurnMs": 7053,
   "inputTokens": 2380,
   "outputTokens": 789,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 915,
   "reasoningTokens": 657,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:57:54.597Z",
   "finishedAt": "2026-10-06T03:58:01.938Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5813,
   "totalMs": 6879,
   "modelTurnMs": 6200,
   "inputTokens": 2375,
   "outputTokens": 576,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 910,
   "reasoningTokens": 444,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:58:01.939Z",
   "finishedAt": "2026-10-06T03:58:14.601Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10892,
   "totalMs": 12136,
   "modelTurnMs": 11431,
   "inputTokens": 4233,
   "outputTokens": 968,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 909,
   "reasoningTokens": 836,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T03:58:14.602Z",
   "finishedAt": "2026-10-06T03:58:22.645Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6355,
   "totalMs": 7461,
   "modelTurnMs": 6648,
   "inputTokens": 2375,
   "outputTokens": 685,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 910,
   "reasoningTokens": 553,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:58:22.647Z",
   "finishedAt": "2026-10-06T03:59:30.190Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 65856,
   "totalMs": 67028,
   "modelTurnMs": 66311,
   "inputTokens": 3879,
   "outputTokens": 8157,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 7994,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 203,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:59:30.191Z",
   "finishedAt": "2026-10-06T03:59:33.148Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1067,
   "totalMs": 2489,
   "modelTurnMs": 1679,
   "inputTokens": 2236,
   "outputTokens": 176,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 771,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:59:33.150Z",
   "finishedAt": "2026-10-06T03:59:38.662Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3517,
   "totalMs": 5056,
   "modelTurnMs": 4219,
   "inputTokens": 2232,
   "outputTokens": 475,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 767,
   "reasoningTokens": 293,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 183,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:59:38.663Z",
   "finishedAt": "2026-10-06T03:59:44.835Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3768,
   "totalMs": 5715,
   "modelTurnMs": 4811,
   "inputTokens": 4091,
   "outputTokens": 448,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 767,
   "reasoningTokens": 266,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 183,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T03:59:44.836Z",
   "finishedAt": "2026-10-06T03:59:49.268Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2523,
   "totalMs": 3980,
   "modelTurnMs": 3169,
   "inputTokens": 2232,
   "outputTokens": 298,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 767,
   "reasoningTokens": 122,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T03:59:49.269Z",
   "finishedAt": "2026-10-06T04:00:06.884Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 15206,
   "totalMs": 17159,
   "modelTurnMs": 16536,
   "inputTokens": 4221,
   "outputTokens": 2637,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 4211,
   "reasoningTokens": 2221,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1237,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T04:00:06.885Z",
   "finishedAt": "2026-10-06T04:00:10.780Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 990,
   "totalMs": 3444,
   "modelTurnMs": 2594,
   "inputTokens": 2668,
   "outputTokens": 450,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1203,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1000,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T04:00:10.781Z",
   "finishedAt": "2026-10-06T04:00:17.442Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2976,
   "totalMs": 6186,
   "modelTurnMs": 5552,
   "inputTokens": 2663,
   "outputTokens": 665,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1198,
   "reasoningTokens": 199,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1026,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T04:00:17.443Z",
   "finishedAt": "2026-10-06T04:00:39.767Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 15391,
   "totalMs": 21853,
   "modelTurnMs": 21196,
   "inputTokens": 4522,
   "outputTokens": 1477,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 1198,
   "reasoningTokens": 948,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1150,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T04:00:39.768Z",
   "finishedAt": "2026-10-06T04:00:51.202Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6210,
   "totalMs": 10952,
   "modelTurnMs": 10194,
   "inputTokens": 2665,
   "outputTokens": 798,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1200,
   "reasoningTokens": 304,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1081,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T04:00:51.203Z",
   "finishedAt": "2026-10-06T04:01:39.514Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 43456,
   "totalMs": 47108,
   "modelTurnMs": 46434,
   "inputTokens": 4023,
   "outputTokens": 6491,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 5934,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "format: output does not start with SELECT or WITH"
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "format: output does not start with SELECT or WITH"
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1420,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "failedOutputLastLine": "```",
   "failedOutputLineCount": 49
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T04:01:39.516Z",
   "finishedAt": "2026-10-06T04:01:47.534Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4480,
   "totalMs": 7560,
   "modelTurnMs": 6863,
   "inputTokens": 2516,
   "outputTokens": 1121,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1051,
   "reasoningTokens": 477,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T04:01:47.535Z",
   "finishedAt": "2026-10-06T04:02:00.989Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7878,
   "totalMs": 12956,
   "modelTurnMs": 12258,
   "inputTokens": 2514,
   "outputTokens": 1545,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1049,
   "reasoningTokens": 808,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1351,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T04:02:00.990Z",
   "finishedAt": "2026-10-06T04:02:26.068Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 17190,
   "totalMs": 24575,
   "modelTurnMs": 23880,
   "inputTokens": 4373,
   "outputTokens": 1860,
   "cacheReadTokens": 3322,
   "cacheWriteTokens": 1049,
   "reasoningTokens": 1091,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1463,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T04:02:26.069Z",
   "finishedAt": "2026-10-06T04:02:39.467Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7840,
   "totalMs": 12910,
   "modelTurnMs": 12157,
   "inputTokens": 2515,
   "outputTokens": 1670,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1050,
   "reasoningTokens": 911,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1321,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T03:23:59.235Z",
   "finishedAt": "2026-10-06T03:23:59.621Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 385,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:23:59.621Z",
   "finishedAt": "2026-10-06T03:23:59.948Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 326,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T03:23:59.948Z",
   "finishedAt": "2026-10-06T03:24:00.310Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 362,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:00.310Z",
   "finishedAt": "2026-10-06T03:24:00.642Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 332,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:00.643Z",
   "finishedAt": "2026-10-06T03:24:01.000Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 357,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:01.000Z",
   "finishedAt": "2026-10-06T03:24:01.394Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 393,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:01.394Z",
   "finishedAt": "2026-10-06T03:24:01.728Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 334,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:01.728Z",
   "finishedAt": "2026-10-06T03:24:02.059Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 330,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:02.059Z",
   "finishedAt": "2026-10-06T03:24:02.393Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 334,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:02.393Z",
   "finishedAt": "2026-10-06T03:24:02.795Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 402,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:02.795Z",
   "finishedAt": "2026-10-06T03:24:03.142Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 347,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:03.142Z",
   "finishedAt": "2026-10-06T03:24:03.464Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 321,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:03.464Z",
   "finishedAt": "2026-10-06T03:24:03.810Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 346,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:03.810Z",
   "finishedAt": "2026-10-06T03:24:04.152Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 341,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:04.152Z",
   "finishedAt": "2026-10-06T03:24:04.485Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 333,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T03:24:04.486Z",
   "finishedAt": "2026-10-06T03:24:04.857Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 371,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:04.857Z",
   "finishedAt": "2026-10-06T03:24:05.243Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 386,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:05.244Z",
   "finishedAt": "2026-10-06T03:24:05.589Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 346,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:05.590Z",
   "finishedAt": "2026-10-06T03:24:05.973Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 383,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:05.974Z",
   "finishedAt": "2026-10-06T03:24:06.296Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 322,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:06.296Z",
   "finishedAt": "2026-10-06T03:24:06.639Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 343,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:06.639Z",
   "finishedAt": "2026-10-06T03:24:06.979Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 339,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:06.979Z",
   "finishedAt": "2026-10-06T03:24:07.381Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 402,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:07.381Z",
   "finishedAt": "2026-10-06T03:24:07.721Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 340,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:07.722Z",
   "finishedAt": "2026-10-06T03:24:08.060Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 338,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:08.060Z",
   "finishedAt": "2026-10-06T03:24:08.422Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 362,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:08.423Z",
   "finishedAt": "2026-10-06T03:24:08.789Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 366,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:08.789Z",
   "finishedAt": "2026-10-06T03:24:09.161Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 371,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:09.161Z",
   "finishedAt": "2026-10-06T03:24:09.576Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 415,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-2",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": null,
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T03:24:09.577Z",
   "finishedAt": "2026-10-06T03:24:09.934Z",
   "status": "blocked",
   "error": "Codex is not signed in with ChatGPT (no account); refusing to run (no API-key fallback).",
   "firstUsefulMs": null,
   "totalMs": 357,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": null,
   "speed": null,
   "outputChars": 0,
   "notes": []
  },
  {
   "batch": "batch-3",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T12:35:37.505Z",
   "finishedAt": "2026-10-06T12:35:50.984Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10262,
   "totalMs": 13198,
   "modelTurnMs": 12522,
   "inputTokens": 12228,
   "outputTokens": 317,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 163,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 479,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T12:35:50.985Z",
   "finishedAt": "2026-10-06T12:36:05.960Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11554,
   "totalMs": 14395,
   "modelTurnMs": 13866,
   "inputTokens": 12230,
   "outputTokens": 355,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 192,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 494,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T13:03:29.833Z",
   "finishedAt": "2026-10-06T13:04:31.814Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 40412,
   "totalMs": 61596,
   "modelTurnMs": 61035,
   "inputTokens": 12255,
   "outputTokens": 1766,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 830,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3584,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T13:04:31.814Z",
   "finishedAt": "2026-10-06T13:05:55.173Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 67998,
   "totalMs": 82988,
   "modelTurnMs": 81609,
   "inputTokens": 12253,
   "outputTokens": 2319,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 1533,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 2932,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T13:05:55.173Z",
   "finishedAt": "2026-10-06T13:06:10.249Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8919,
   "totalMs": 14792,
   "modelTurnMs": 13554,
   "inputTokens": 12260,
   "outputTokens": 445,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 86,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1292,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T13:06:10.250Z",
   "finishedAt": "2026-10-06T13:06:33.435Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 14869,
   "totalMs": 22905,
   "modelTurnMs": 22369,
   "inputTokens": 12258,
   "outputTokens": 606,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 256,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1346,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T13:06:33.435Z",
   "finishedAt": "2026-10-06T13:06:41.972Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7930,
   "totalMs": 8535,
   "modelTurnMs": 8071,
   "inputTokens": 12256,
   "outputTokens": 237,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 199,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 47,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T13:06:41.973Z",
   "finishedAt": "2026-10-06T13:06:57.415Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 14810,
   "totalMs": 15440,
   "modelTurnMs": 14934,
   "inputTokens": 12256,
   "outputTokens": 385,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 347,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 47,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T13:06:57.416Z",
   "finishedAt": "2026-10-06T13:07:10.351Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11084,
   "totalMs": 12634,
   "modelTurnMs": 12160,
   "inputTokens": 12350,
   "outputTokens": 284,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 189,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 217,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T13:07:10.351Z",
   "finishedAt": "2026-10-06T13:07:23.813Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11231,
   "totalMs": 13047,
   "modelTurnMs": 12491,
   "inputTokens": 12352,
   "outputTokens": 284,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 189,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 217,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T13:07:23.814Z",
   "finishedAt": "2026-10-06T13:07:33.734Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6538,
   "totalMs": 9632,
   "modelTurnMs": 9085,
   "inputTokens": 12202,
   "outputTokens": 261,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 101,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 227,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T13:07:33.734Z",
   "finishedAt": "2026-10-06T13:07:51.859Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 15074,
   "totalMs": 17843,
   "modelTurnMs": 17303,
   "inputTokens": 12200,
   "outputTokens": 458,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 306,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 213,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T13:07:51.859Z",
   "finishedAt": "2026-10-06T13:08:03.876Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7188,
   "totalMs": 11735,
   "modelTurnMs": 11248,
   "inputTokens": 12492,
   "outputTokens": 346,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 89,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 872,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T13:08:03.876Z",
   "finishedAt": "2026-10-06T13:08:23.560Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 14482,
   "totalMs": 19388,
   "modelTurnMs": 18900,
   "inputTokens": 12492,
   "outputTokens": 496,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 224,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 955,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T13:08:23.561Z",
   "finishedAt": "2026-10-06T13:08:37.639Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6090,
   "totalMs": 13764,
   "modelTurnMs": 13265,
   "inputTokens": 12335,
   "outputTokens": 534,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 61,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1678,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T13:08:37.640Z",
   "finishedAt": "2026-10-06T13:08:56.339Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9993,
   "totalMs": 18402,
   "modelTurnMs": 17804,
   "inputTokens": 12333,
   "outputTokens": 620,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 181,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1500,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T13:08:56.339Z",
   "finishedAt": "2026-10-06T13:09:11.258Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11636,
   "totalMs": 14627,
   "modelTurnMs": 14109,
   "inputTokens": 12228,
   "outputTokens": 300,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 137,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 495,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T13:09:11.259Z",
   "finishedAt": "2026-10-06T13:09:26.668Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12149,
   "totalMs": 15121,
   "modelTurnMs": 14647,
   "inputTokens": 12232,
   "outputTokens": 389,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 226,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 494,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T13:09:26.668Z",
   "finishedAt": "2026-10-06T13:10:23.405Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 39695,
   "totalMs": 56393,
   "modelTurnMs": 55059,
   "inputTokens": 12255,
   "outputTokens": 1674,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 839,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3112,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T13:10:23.405Z",
   "finishedAt": "2026-10-06T13:11:55.922Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 75913,
   "totalMs": 92205,
   "modelTurnMs": 91691,
   "inputTokens": 12249,
   "outputTokens": 2569,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 1750,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3045,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T13:11:55.923Z",
   "finishedAt": "2026-10-06T13:12:12.936Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10855,
   "totalMs": 16638,
   "modelTurnMs": 16156,
   "inputTokens": 12260,
   "outputTokens": 474,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 112,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1327,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T13:12:12.936Z",
   "finishedAt": "2026-10-06T13:12:33.797Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13228,
   "totalMs": 20564,
   "modelTurnMs": 19887,
   "inputTokens": 12258,
   "outputTokens": 583,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 220,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1328,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T13:12:33.798Z",
   "finishedAt": "2026-10-06T13:12:46.302Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11428,
   "totalMs": 12504,
   "modelTurnMs": 12029,
   "inputTokens": 12258,
   "outputTokens": 289,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 251,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 47,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T13:12:46.303Z",
   "finishedAt": "2026-10-06T13:13:08.811Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 18482,
   "totalMs": 22508,
   "modelTurnMs": 22026,
   "inputTokens": 12256,
   "outputTokens": 413,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 375,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 47,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T13:13:08.811Z",
   "finishedAt": "2026-10-06T13:13:20.981Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10207,
   "totalMs": 11808,
   "modelTurnMs": 11262,
   "inputTokens": 12352,
   "outputTokens": 292,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 197,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 217,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T13:13:20.981Z",
   "finishedAt": "2026-10-06T13:13:33.460Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10805,
   "totalMs": 12209,
   "modelTurnMs": 11725,
   "inputTokens": 12352,
   "outputTokens": 324,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 229,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 217,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T13:13:33.461Z",
   "finishedAt": "2026-10-06T13:13:46.789Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10488,
   "totalMs": 13013,
   "modelTurnMs": 11676,
   "inputTokens": 12202,
   "outputTokens": 342,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 193,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 207,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T13:13:46.789Z",
   "finishedAt": "2026-10-06T13:13:58.729Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8934,
   "totalMs": 11671,
   "modelTurnMs": 10521,
   "inputTokens": 12200,
   "outputTokens": 338,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 189,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 207,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T13:13:58.730Z",
   "finishedAt": "2026-10-06T13:14:10.864Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7157,
   "totalMs": 11852,
   "modelTurnMs": 11161,
   "inputTokens": 12492,
   "outputTokens": 328,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 71,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 860,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T13:14:10.864Z",
   "finishedAt": "2026-10-06T13:14:26.566Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10741,
   "totalMs": 15428,
   "modelTurnMs": 14956,
   "inputTokens": 12492,
   "outputTokens": 401,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 144,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 860,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T13:14:26.566Z",
   "finishedAt": "2026-10-06T13:14:44.634Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8616,
   "totalMs": 17809,
   "modelTurnMs": 17333,
   "inputTokens": 12335,
   "outputTokens": 588,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 119,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1654,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T13:14:44.634Z",
   "finishedAt": "2026-10-06T13:15:04.543Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11072,
   "totalMs": 19620,
   "modelTurnMs": 18972,
   "inputTokens": 12339,
   "outputTokens": 691,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 222,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1634,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  }
 ],
 "protocolNote": "The run protocol was declared before the first run. Its public summary is the Method section of the study page."
}
