{
 "note": "Receipts copied from public-runs/effort-ladder. Every attempt is kept, failures included.",
 "cases": [
  {
   "id": "merge-ranges",
   "title": "Fix an interval-merge function (off-by-one and edge cases)",
   "kind": "code-fix",
   "difficulty": "hard",
   "format": "js",
   "validator": "sandbox"
  },
  {
   "id": "day-hours",
   "title": "Fix a time-zone day-length function (DST)",
   "kind": "code-fix",
   "difficulty": "hard",
   "format": "js",
   "validator": "sandbox"
  },
  {
   "id": "csv-parse",
   "title": "Write a CSV parser (quoted newlines, strict errors)",
   "kind": "code-write",
   "difficulty": "hard",
   "format": "js",
   "validator": "sandbox"
  },
  {
   "id": "event-loop-order",
   "title": "Predict JavaScript event-loop output order",
   "kind": "reasoning",
   "difficulty": "hard",
   "format": "line",
   "validator": "exact",
   "expected": "A1,S,M1,A2,Q1,TH,M3,M2,Q2,B1,M4,R1,T1,T2,T2M,A3"
  },
  {
   "id": "talk-schedule",
   "title": "Solve a multi-constraint room schedule",
   "kind": "constraint",
   "difficulty": "hard",
   "format": "json",
   "validator": "sandbox"
  },
  {
   "id": "semver-regex",
   "title": "Write a strict SemVer 2.0.0 regex",
   "kind": "regex",
   "difficulty": "hard",
   "format": "regex",
   "validator": "sandbox"
  },
  {
   "id": "money-refactor",
   "title": "Refactor to remove duplication, keep 20 tests green",
   "kind": "refactor",
   "difficulty": "hard",
   "format": "js",
   "validator": "sandbox"
  },
  {
   "id": "q1-sql",
   "title": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "kind": "sql",
   "difficulty": "hard",
   "format": "sql",
   "validator": "sandbox"
  }
 ],
 "controls": {
  "ranAt": "2026-10-06T14:21:50.803Z",
  "allOk": true,
  "cases": [
   {
    "caseId": "merge-ranges",
    "referencePassed": true,
    "referenceChecks": 18,
    "wrongAnswers": 5,
    "wrongAnswersFailed": 5,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "day-hours",
    "referencePassed": true,
    "referenceChecks": 36,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "csv-parse",
    "referencePassed": true,
    "referenceChecks": 22,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "event-loop-order",
    "referencePassed": true,
    "referenceChecks": 1,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "talk-schedule",
    "referencePassed": true,
    "referenceChecks": 14,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "semver-regex",
    "referencePassed": true,
    "referenceChecks": 30,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "money-refactor",
    "referencePassed": true,
    "referenceChecks": 25,
    "wrongAnswers": 2,
    "wrongAnswersFailed": 2,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "q1-sql",
    "referencePassed": true,
    "referenceChecks": 8,
    "wrongAnswers": 4,
    "wrongAnswersFailed": 4,
    "wrappedReferenceIsFormatMiss": true
   }
  ]
 },
 "references": {
  "group": "provider-h2h-hard",
  "cells": [
   "claude-cli sonnet default (reps 1-2)",
   "claude-cli opus default (reps 1-2)",
   "claude-cli opus high (reps 1-2)",
   "codex-cli gpt-6.1-sol medium",
   "codex-cli gpt-6.1-sol high"
  ]
 },
 "batches": [
  {
   "batch": "batch-1",
   "schema": "agent-provider-h2h@2",
   "route": "claude-cli",
   "startedAt": "2026-10-06T14:21:50.938Z",
   "finishedAt": "2026-10-06T14:34:45.256Z",
   "stopReason": null,
   "trimmed": [],
   "generatedAt": null
  },
  {
   "batch": "batch-2",
   "schema": "agent-provider-h2h@2",
   "route": "codex-cli",
   "startedAt": "2026-10-06T14:20:26.190Z",
   "finishedAt": "2026-10-06T14:24:55.305Z",
   "stopReason": null,
   "trimmed": [],
   "generatedAt": null
  }
 ],
 "receipts": [
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:21:50.938Z",
   "finishedAt": "2026-10-06T14:21:54.198Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1298,
   "totalMs": 2786,
   "modelTurnMs": 1888,
   "inputTokens": 2235,
   "outputTokens": 224,
   "cacheReadTokens": 531,
   "cacheWriteTokens": 1702,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 461,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:21:54.199Z",
   "finishedAt": "2026-10-06T14:21:57.384Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1220,
   "totalMs": 2712,
   "modelTurnMs": 1767,
   "inputTokens": 2234,
   "outputTokens": 224,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 769,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 461,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T14:21:57.385Z",
   "finishedAt": "2026-10-06T14:22:00.780Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1333,
   "totalMs": 2932,
   "modelTurnMs": 2077,
   "inputTokens": 2236,
   "outputTokens": 224,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 771,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 461,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:22:00.780Z",
   "finishedAt": "2026-10-06T14:22:04.587Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1507,
   "totalMs": 3343,
   "modelTurnMs": 2435,
   "inputTokens": 2231,
   "outputTokens": 220,
   "cacheReadTokens": 531,
   "cacheWriteTokens": 1698,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 451,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:22:04.587Z",
   "finishedAt": "2026-10-06T14:22:10.541Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3547,
   "totalMs": 5509,
   "modelTurnMs": 4658,
   "inputTokens": 2231,
   "outputTokens": 343,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 766,
   "reasoningTokens": 109,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 456,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:22:10.541Z",
   "finishedAt": "2026-10-06T14:22:30.957Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 15602,
   "totalMs": 19960,
   "modelTurnMs": 19335,
   "inputTokens": 2261,
   "outputTokens": 2263,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 796,
   "reasoningTokens": 1489,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1767,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:22:30.957Z",
   "finishedAt": "2026-10-06T14:22:53.943Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 19452,
   "totalMs": 22480,
   "modelTurnMs": 21789,
   "inputTokens": 2260,
   "outputTokens": 2574,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 795,
   "reasoningTokens": 1941,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1336,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T14:22:53.943Z",
   "finishedAt": "2026-10-06T14:23:21.582Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 24592,
   "totalMs": 27179,
   "modelTurnMs": 26449,
   "inputTokens": 2261,
   "outputTokens": 3106,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 796,
   "reasoningTokens": 2597,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1074,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:23:21.582Z",
   "finishedAt": "2026-10-06T14:23:35.060Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5710,
   "totalMs": 13020,
   "modelTurnMs": 12046,
   "inputTokens": 2256,
   "outputTokens": 1349,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 791,
   "reasoningTokens": 394,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1900,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:23:35.060Z",
   "finishedAt": "2026-10-06T14:24:06.637Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 24590,
   "totalMs": 31116,
   "modelTurnMs": 30477,
   "inputTokens": 2255,
   "outputTokens": 2913,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 790,
   "reasoningTokens": 1993,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1878,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:24:06.638Z",
   "finishedAt": "2026-10-06T14:24:11.408Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1518,
   "totalMs": 4324,
   "modelTurnMs": 3476,
   "inputTokens": 2272,
   "outputTokens": 553,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 807,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1532,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:24:11.409Z",
   "finishedAt": "2026-10-06T14:24:19.704Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5174,
   "totalMs": 7826,
   "modelTurnMs": 7135,
   "inputTokens": 2273,
   "outputTokens": 954,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 808,
   "reasoningTokens": 423,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1434,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T14:24:19.704Z",
   "finishedAt": "2026-10-06T14:24:30.180Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7606,
   "totalMs": 10007,
   "modelTurnMs": 9231,
   "inputTokens": 2271,
   "outputTokens": 1257,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 806,
   "reasoningTokens": 770,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1270,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:24:30.180Z",
   "finishedAt": "2026-10-06T14:24:40.504Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6269,
   "totalMs": 9879,
   "modelTurnMs": 9031,
   "inputTokens": 2267,
   "outputTokens": 984,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 802,
   "reasoningTokens": 471,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1293,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:24:40.504Z",
   "finishedAt": "2026-10-06T14:24:59.212Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 14597,
   "totalMs": 18264,
   "modelTurnMs": 17583,
   "inputTokens": 2268,
   "outputTokens": 981,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 803,
   "reasoningTokens": 444,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1365,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:24:59.212Z",
   "finishedAt": "2026-10-06T14:25:08.203Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8365,
   "totalMs": 8803,
   "modelTurnMs": 8114,
   "inputTokens": 2291,
   "outputTokens": 1046,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 826,
   "reasoningTokens": 995,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:25:08.204Z",
   "finishedAt": "2026-10-06T14:25:18.383Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9547,
   "totalMs": 9975,
   "modelTurnMs": 9298,
   "inputTokens": 2290,
   "outputTokens": 1251,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 825,
   "reasoningTokens": 1200,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T14:25:18.383Z",
   "finishedAt": "2026-10-06T14:25:29.501Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10403,
   "totalMs": 10892,
   "modelTurnMs": 10170,
   "inputTokens": 2290,
   "outputTokens": 1346,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 825,
   "reasoningTokens": 1295,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:25:29.502Z",
   "finishedAt": "2026-10-06T14:25:38.428Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8064,
   "totalMs": 8700,
   "modelTurnMs": 7897,
   "inputTokens": 2286,
   "outputTokens": 735,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 821,
   "reasoningTokens": 684,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:25:38.429Z",
   "finishedAt": "2026-10-06T14:25:51.062Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11867,
   "totalMs": 12422,
   "modelTurnMs": 11607,
   "inputTokens": 2285,
   "outputTokens": 1171,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 820,
   "reasoningTokens": 1120,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:25:51.063Z",
   "finishedAt": "2026-10-06T14:25:57.994Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5687,
   "totalMs": 6491,
   "modelTurnMs": 5779,
   "inputTokens": 2380,
   "outputTokens": 730,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 915,
   "reasoningTokens": 598,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:25:57.994Z",
   "finishedAt": "2026-10-06T14:26:06.320Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7084,
   "totalMs": 7899,
   "modelTurnMs": 7074,
   "inputTokens": 2379,
   "outputTokens": 783,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 914,
   "reasoningTokens": 651,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T14:26:06.321Z",
   "finishedAt": "2026-10-06T14:26:15.816Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8301,
   "totalMs": 9066,
   "modelTurnMs": 8341,
   "inputTokens": 2379,
   "outputTokens": 866,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 914,
   "reasoningTokens": 734,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:26:15.817Z",
   "finishedAt": "2026-10-06T14:26:23.986Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6586,
   "totalMs": 7718,
   "modelTurnMs": 6892,
   "inputTokens": 2375,
   "outputTokens": 630,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 910,
   "reasoningTokens": 498,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:26:23.986Z",
   "finishedAt": "2026-10-06T14:26:31.406Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5822,
   "totalMs": 6986,
   "modelTurnMs": 6238,
   "inputTokens": 2374,
   "outputTokens": 603,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 909,
   "reasoningTokens": 471,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:26:31.406Z",
   "finishedAt": "2026-10-06T14:26:35.343Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1986,
   "totalMs": 3511,
   "modelTurnMs": 2589,
   "inputTokens": 2236,
   "outputTokens": 176,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 771,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:26:35.343Z",
   "finishedAt": "2026-10-06T14:26:39.882Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2951,
   "totalMs": 4117,
   "modelTurnMs": 3075,
   "inputTokens": 2236,
   "outputTokens": 433,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 771,
   "reasoningTokens": 251,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T14:26:39.882Z",
   "finishedAt": "2026-10-06T14:26:44.318Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2768,
   "totalMs": 4013,
   "modelTurnMs": 3165,
   "inputTokens": 2236,
   "outputTokens": 434,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 771,
   "reasoningTokens": 252,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:26:44.318Z",
   "finishedAt": "2026-10-06T14:26:48.957Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3806,
   "totalMs": 4206,
   "modelTurnMs": 3322,
   "inputTokens": 2232,
   "outputTokens": 270,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 767,
   "reasoningTokens": 88,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:26:48.958Z",
   "finishedAt": "2026-10-06T14:26:56.233Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4786,
   "totalMs": 6799,
   "modelTurnMs": 5959,
   "inputTokens": 2232,
   "outputTokens": 301,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 767,
   "reasoningTokens": 119,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 181,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:26:56.233Z",
   "finishedAt": "2026-10-06T14:27:00.525Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1437,
   "totalMs": 3715,
   "modelTurnMs": 2807,
   "inputTokens": 2669,
   "outputTokens": 440,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1204,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 914,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:27:00.525Z",
   "finishedAt": "2026-10-06T14:27:04.884Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1322,
   "totalMs": 3888,
   "modelTurnMs": 2979,
   "inputTokens": 2669,
   "outputTokens": 466,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1204,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1047,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T14:27:04.885Z",
   "finishedAt": "2026-10-06T14:27:13.463Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5896,
   "totalMs": 8089,
   "modelTurnMs": 7405,
   "inputTokens": 2668,
   "outputTokens": 1126,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1203,
   "reasoningTokens": 693,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 945,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:27:13.464Z",
   "finishedAt": "2026-10-06T14:27:19.562Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2197,
   "totalMs": 5646,
   "modelTurnMs": 4684,
   "inputTokens": 2664,
   "outputTokens": 448,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1199,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 992,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:27:19.562Z",
   "finishedAt": "2026-10-06T14:27:26.870Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3533,
   "totalMs": 6873,
   "modelTurnMs": 6108,
   "inputTokens": 2664,
   "outputTokens": 660,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1199,
   "reasoningTokens": 183,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1049,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:27:26.871Z",
   "finishedAt": "2026-10-06T14:27:35.229Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4526,
   "totalMs": 7916,
   "modelTurnMs": 7182,
   "inputTokens": 2518,
   "outputTokens": 1272,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1053,
   "reasoningTokens": 545,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1378,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:27:35.230Z",
   "finishedAt": "2026-10-06T14:27:43.119Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4214,
   "totalMs": 7427,
   "modelTurnMs": 6665,
   "inputTokens": 2517,
   "outputTokens": 1076,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1052,
   "reasoningTokens": 421,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1210,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-06T14:27:43.120Z",
   "finishedAt": "2026-10-06T14:27:54.372Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7357,
   "totalMs": 10789,
   "modelTurnMs": 10036,
   "inputTokens": 2518,
   "outputTokens": 1572,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1053,
   "reasoningTokens": 826,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1355,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:27:54.372Z",
   "finishedAt": "2026-10-06T14:28:05.437Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2653,
   "totalMs": 10644,
   "modelTurnMs": 9960,
   "inputTokens": 2513,
   "outputTokens": 779,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1048,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1402,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:28:05.437Z",
   "finishedAt": "2026-10-06T14:28:19.014Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7748,
   "totalMs": 13134,
   "modelTurnMs": 12353,
   "inputTokens": 2514,
   "outputTokens": 1571,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1049,
   "reasoningTokens": 793,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1399,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:28:19.014Z",
   "finishedAt": "2026-10-06T14:28:22.279Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1307,
   "totalMs": 2784,
   "modelTurnMs": 1839,
   "inputTokens": 2233,
   "outputTokens": 224,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 768,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 461,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:28:22.280Z",
   "finishedAt": "2026-10-06T14:28:25.663Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1395,
   "totalMs": 2928,
   "modelTurnMs": 2049,
   "inputTokens": 2234,
   "outputTokens": 224,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 769,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 461,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T14:28:25.664Z",
   "finishedAt": "2026-10-06T14:28:29.567Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2112,
   "totalMs": 3485,
   "modelTurnMs": 2593,
   "inputTokens": 2234,
   "outputTokens": 220,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 769,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 451,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:28:29.567Z",
   "finishedAt": "2026-10-06T14:28:33.441Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1611,
   "totalMs": 3441,
   "modelTurnMs": 2515,
   "inputTokens": 2231,
   "outputTokens": 220,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 766,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 456,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:28:33.442Z",
   "finishedAt": "2026-10-06T14:28:39.156Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3726,
   "totalMs": 5294,
   "modelTurnMs": 4387,
   "inputTokens": 2231,
   "outputTokens": 337,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 766,
   "reasoningTokens": 115,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 431,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:28:39.157Z",
   "finishedAt": "2026-10-06T14:28:55.895Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12332,
   "totalMs": 16280,
   "modelTurnMs": 15459,
   "inputTokens": 2260,
   "outputTokens": 1832,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 795,
   "reasoningTokens": 1091,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1556,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:28:55.896Z",
   "finishedAt": "2026-10-06T14:29:20.356Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 20330,
   "totalMs": 24010,
   "modelTurnMs": 23316,
   "inputTokens": 2260,
   "outputTokens": 2836,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 795,
   "reasoningTokens": 2093,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1598,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T14:29:20.357Z",
   "finishedAt": "2026-10-06T14:29:56.609Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 33000,
   "totalMs": 35813,
   "modelTurnMs": 35172,
   "inputTokens": 2260,
   "outputTokens": 4187,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 795,
   "reasoningTokens": 3610,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1256,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:29:56.609Z",
   "finishedAt": "2026-10-06T14:30:12.882Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8871,
   "totalMs": 15822,
   "modelTurnMs": 15156,
   "inputTokens": 2257,
   "outputTokens": 1569,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 792,
   "reasoningTokens": 587,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 2025,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:30:12.883Z",
   "finishedAt": "2026-10-06T14:30:44.758Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 22628,
   "totalMs": 31359,
   "modelTurnMs": 30665,
   "inputTokens": 2256,
   "outputTokens": 3075,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 791,
   "reasoningTokens": 1846,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 2877,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:30:44.759Z",
   "finishedAt": "2026-10-06T14:30:49.591Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1241,
   "totalMs": 4397,
   "modelTurnMs": 3507,
   "inputTokens": 2272,
   "outputTokens": 603,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 807,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1686,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:30:49.591Z",
   "finishedAt": "2026-10-06T14:30:54.669Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1535,
   "totalMs": 4646,
   "modelTurnMs": 3605,
   "inputTokens": 2272,
   "outputTokens": 561,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 807,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1552,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T14:30:54.669Z",
   "finishedAt": "2026-10-06T14:31:11.215Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13680,
   "totalMs": 16100,
   "modelTurnMs": 15363,
   "inputTokens": 2272,
   "outputTokens": 1276,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 807,
   "reasoningTokens": 783,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1339,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:31:11.216Z",
   "finishedAt": "2026-10-06T14:31:17.871Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2230,
   "totalMs": 6214,
   "modelTurnMs": 5456,
   "inputTokens": 2268,
   "outputTokens": 531,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 803,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1368,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:31:17.871Z",
   "finishedAt": "2026-10-06T14:31:29.192Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7304,
   "totalMs": 10909,
   "modelTurnMs": 10181,
   "inputTokens": 2268,
   "outputTokens": 1178,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 803,
   "reasoningTokens": 655,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1259,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:31:29.193Z",
   "finishedAt": "2026-10-06T14:31:39.112Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9084,
   "totalMs": 9728,
   "modelTurnMs": 8860,
   "inputTokens": 2291,
   "outputTokens": 1025,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 826,
   "reasoningTokens": 974,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:31:39.113Z",
   "finishedAt": "2026-10-06T14:31:48.040Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8240,
   "totalMs": 8723,
   "modelTurnMs": 7996,
   "inputTokens": 2290,
   "outputTokens": 1149,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 825,
   "reasoningTokens": 1098,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T14:31:48.041Z",
   "finishedAt": "2026-10-06T14:31:59.037Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10241,
   "totalMs": 10773,
   "modelTurnMs": 9976,
   "inputTokens": 2290,
   "outputTokens": 1277,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 825,
   "reasoningTokens": 1226,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:31:59.038Z",
   "finishedAt": "2026-10-06T14:32:08.404Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8495,
   "totalMs": 9162,
   "modelTurnMs": 8335,
   "inputTokens": 2286,
   "outputTokens": 842,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 821,
   "reasoningTokens": 791,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:32:08.405Z",
   "finishedAt": "2026-10-06T14:32:19.809Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10615,
   "totalMs": 11191,
   "modelTurnMs": 10467,
   "inputTokens": 2285,
   "outputTokens": 1065,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 820,
   "reasoningTokens": 1014,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 47,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:32:19.811Z",
   "finishedAt": "2026-10-06T14:32:26.953Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5916,
   "totalMs": 6712,
   "modelTurnMs": 6008,
   "inputTokens": 2380,
   "outputTokens": 756,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 915,
   "reasoningTokens": 624,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:32:26.954Z",
   "finishedAt": "2026-10-06T14:32:37.010Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8756,
   "totalMs": 9610,
   "modelTurnMs": 8869,
   "inputTokens": 2379,
   "outputTokens": 756,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 914,
   "reasoningTokens": 624,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T14:32:37.010Z",
   "finishedAt": "2026-10-06T14:32:45.545Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7312,
   "totalMs": 8096,
   "modelTurnMs": 7426,
   "inputTokens": 2379,
   "outputTokens": 887,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 914,
   "reasoningTokens": 755,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:32:45.546Z",
   "finishedAt": "2026-10-06T14:32:53.246Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6165,
   "totalMs": 7288,
   "modelTurnMs": 6491,
   "inputTokens": 2376,
   "outputTokens": 558,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 911,
   "reasoningTokens": 426,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:32:53.247Z",
   "finishedAt": "2026-10-06T14:33:02.213Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7408,
   "totalMs": 8532,
   "modelTurnMs": 7781,
   "inputTokens": 2374,
   "outputTokens": 697,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 909,
   "reasoningTokens": 565,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 217,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:33:02.214Z",
   "finishedAt": "2026-10-06T14:33:05.543Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1569,
   "totalMs": 2855,
   "modelTurnMs": 2010,
   "inputTokens": 2236,
   "outputTokens": 176,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 771,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:33:05.544Z",
   "finishedAt": "2026-10-06T14:33:09.804Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2642,
   "totalMs": 3816,
   "modelTurnMs": 2874,
   "inputTokens": 2236,
   "outputTokens": 421,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 771,
   "reasoningTokens": 245,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T14:33:09.805Z",
   "finishedAt": "2026-10-06T14:33:13.960Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2545,
   "totalMs": 3714,
   "modelTurnMs": 2835,
   "inputTokens": 2236,
   "outputTokens": 425,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 771,
   "reasoningTokens": 243,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 177,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:33:13.961Z",
   "finishedAt": "2026-10-06T14:33:18.212Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3171,
   "totalMs": 3786,
   "modelTurnMs": 2840,
   "inputTokens": 2232,
   "outputTokens": 268,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 767,
   "reasoningTokens": 86,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 183,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:33:18.213Z",
   "finishedAt": "2026-10-06T14:33:23.483Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3227,
   "totalMs": 4780,
   "modelTurnMs": 3904,
   "inputTokens": 2233,
   "outputTokens": 425,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 768,
   "reasoningTokens": 243,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 181,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:33:23.484Z",
   "finishedAt": "2026-10-06T14:33:29.104Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2084,
   "totalMs": 5156,
   "modelTurnMs": 4251,
   "inputTokens": 2667,
   "outputTokens": 429,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1202,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 923,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:33:29.105Z",
   "finishedAt": "2026-10-06T14:33:33.513Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1445,
   "totalMs": 3991,
   "modelTurnMs": 3179,
   "inputTokens": 2667,
   "outputTokens": 490,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1202,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1106,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T14:33:33.513Z",
   "finishedAt": "2026-10-06T14:33:41.546Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5300,
   "totalMs": 7608,
   "modelTurnMs": 6733,
   "inputTokens": 2668,
   "outputTokens": 1018,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1203,
   "reasoningTokens": 580,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 939,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:33:41.547Z",
   "finishedAt": "2026-10-06T14:33:46.914Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1515,
   "totalMs": 4934,
   "modelTurnMs": 4035,
   "inputTokens": 2663,
   "outputTokens": 459,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1198,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1025,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:33:46.915Z",
   "finishedAt": "2026-10-06T14:33:54.534Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3695,
   "totalMs": 7074,
   "modelTurnMs": 6273,
   "inputTokens": 2664,
   "outputTokens": 725,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1199,
   "reasoningTokens": 253,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1041,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:33:54.535Z",
   "finishedAt": "2026-10-06T14:34:02.722Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4693,
   "totalMs": 7683,
   "modelTurnMs": 6845,
   "inputTokens": 2518,
   "outputTokens": 1217,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1053,
   "reasoningTokens": 591,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1138,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:34:02.722Z",
   "finishedAt": "2026-10-06T14:34:12.379Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6067,
   "totalMs": 9184,
   "modelTurnMs": 8494,
   "inputTokens": 2519,
   "outputTokens": 1250,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1054,
   "reasoningTokens": 566,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1236,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-06T14:34:12.380Z",
   "finishedAt": "2026-10-06T14:34:21.369Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5200,
   "totalMs": 8546,
   "modelTurnMs": 7829,
   "inputTokens": 2519,
   "outputTokens": 1322,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1054,
   "reasoningTokens": 626,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1280,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:34:21.370Z",
   "finishedAt": "2026-10-06T14:34:31.019Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2512,
   "totalMs": 9226,
   "modelTurnMs": 8523,
   "inputTokens": 2513,
   "outputTokens": 772,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1048,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1417,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:34:31.020Z",
   "finishedAt": "2026-10-06T14:34:45.255Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8542,
   "totalMs": 13804,
   "modelTurnMs": 13005,
   "inputTokens": 2514,
   "outputTokens": 1611,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1049,
   "reasoningTokens": 829,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1436,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:20:26.190Z",
   "finishedAt": "2026-10-06T14:20:43.634Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13078,
   "totalMs": 17196,
   "modelTurnMs": 15817,
   "inputTokens": 12228,
   "outputTokens": 238,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 63,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 520,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:20:43.634Z",
   "finishedAt": "2026-10-06T14:21:28.214Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 27541,
   "totalMs": 44291,
   "modelTurnMs": 43772,
   "inputTokens": 12251,
   "outputTokens": 1310,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 458,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3285,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:21:28.214Z",
   "finishedAt": "2026-10-06T14:21:46.104Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9595,
   "totalMs": 17659,
   "modelTurnMs": 16401,
   "inputTokens": 12262,
   "outputTokens": 440,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 78,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1324,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:21:46.104Z",
   "finishedAt": "2026-10-06T14:22:00.712Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13643,
   "totalMs": 14607,
   "modelTurnMs": 14136,
   "inputTokens": 12256,
   "outputTokens": 236,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 198,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 47,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:22:00.712Z",
   "finishedAt": "2026-10-06T14:22:14.575Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11996,
   "totalMs": 13602,
   "modelTurnMs": 12391,
   "inputTokens": 12352,
   "outputTokens": 284,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 189,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 217,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:22:14.575Z",
   "finishedAt": "2026-10-06T14:22:23.389Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5521,
   "totalMs": 8574,
   "modelTurnMs": 8098,
   "inputTokens": 12202,
   "outputTokens": 201,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 44,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 221,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:22:23.389Z",
   "finishedAt": "2026-10-06T14:22:36.748Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7743,
   "totalMs": 13115,
   "modelTurnMs": 12606,
   "inputTokens": 12492,
   "outputTokens": 321,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 62,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 866,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-06T14:22:36.749Z",
   "finishedAt": "2026-10-06T14:22:50.643Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4297,
   "totalMs": 13638,
   "modelTurnMs": 13081,
   "inputTokens": 12337,
   "outputTokens": 473,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1716,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "merge-ranges",
   "caseTitle": "Fix an interval-merge function (off-by-one and edge cases)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:22:50.643Z",
   "finishedAt": "2026-10-06T14:22:58.815Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3931,
   "totalMs": 7937,
   "modelTurnMs": 7432,
   "inputTokens": 12228,
   "outputTokens": 172,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 18,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 18,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 523,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "day-hours",
   "caseTitle": "Fix a time-zone day-length function (DST)",
   "caseKind": "code-fix",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:22:58.816Z",
   "finishedAt": "2026-10-06T14:23:39.825Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 23230,
   "totalMs": 40706,
   "modelTurnMs": 40186,
   "inputTokens": 12253,
   "outputTokens": 1225,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 400,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 36,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 36,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3000,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "csv-parse",
   "caseTitle": "Write a CSV parser (quoted newlines, strict errors)",
   "caseKind": "code-write",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:23:39.825Z",
   "finishedAt": "2026-10-06T14:23:57.932Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3789,
   "totalMs": 17847,
   "modelTurnMs": 17334,
   "inputTokens": 12260,
   "outputTokens": 357,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 22,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 22,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1308,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "event-loop-order",
   "caseTitle": "Predict JavaScript event-loop output order",
   "caseKind": "reasoning",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:23:57.932Z",
   "finishedAt": "2026-10-06T14:24:09.030Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10390,
   "totalMs": 11098,
   "modelTurnMs": 10576,
   "inputTokens": 12256,
   "outputTokens": 223,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 185,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 47,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "talk-schedule",
   "caseTitle": "Solve a multi-constraint room schedule",
   "caseKind": "constraint",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:24:09.031Z",
   "finishedAt": "2026-10-06T14:24:21.701Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10763,
   "totalMs": 12410,
   "modelTurnMs": 11847,
   "inputTokens": 12352,
   "outputTokens": 284,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 189,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 14,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 217,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "semver-regex",
   "caseTitle": "Write a strict SemVer 2.0.0 regex",
   "caseKind": "regex",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:24:21.701Z",
   "finishedAt": "2026-10-06T14:24:30.846Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6255,
   "totalMs": 8903,
   "modelTurnMs": 8296,
   "inputTokens": 12200,
   "outputTokens": 199,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 50,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 30,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 30,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 207,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "money-refactor",
   "caseTitle": "Refactor to remove duplication, keep 20 tests green",
   "caseKind": "refactor",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:24:30.847Z",
   "finishedAt": "2026-10-06T14:24:40.628Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4417,
   "totalMs": 9542,
   "modelTurnMs": 9006,
   "inputTokens": 12492,
   "outputTokens": 254,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 25,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 25,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 869,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "q1-sql",
   "caseTitle": "Write a SQLite reporting query (fan-out, ties, boundaries)",
   "caseKind": "sql",
   "caseDifficulty": "hard",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-06T14:24:40.628Z",
   "finishedAt": "2026-10-06T14:24:55.305Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5822,
   "totalMs": 14422,
   "modelTurnMs": 13760,
   "inputTokens": 12335,
   "outputTokens": 507,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 40,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 8,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 8,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 1632,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  }
 ],
 "protocolNote": "The run protocol was declared before the first run. Its public summary is the Method section of the study page."
}
