{
 "note": "Receipts copied from the run folder of the harder-tasks study. Every attempt is kept, failures included. Replies and expected answers are not copied.",
 "cases": [
  {
   "id": "nonogram",
   "title": "Solve a 10x10 nonogram (one solution)",
   "kind": "reasoning",
   "difficulty": "harder",
   "format": "nonogram",
   "validator": "exact",
   "checks": 1
  },
  {
   "id": "sudoku",
   "title": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "kind": "reasoning",
   "difficulty": "harder",
   "format": "sudoku",
   "validator": "exact",
   "checks": 1
  },
  {
   "id": "skyscrapers",
   "title": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "kind": "reasoning",
   "difficulty": "harder",
   "format": "grid",
   "validator": "exact",
   "checks": 1
  },
  {
   "id": "lcg-shuffle",
   "title": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "kind": "code-reading",
   "difficulty": "harder",
   "format": "line",
   "validator": "exact",
   "checks": 1
  }
 ],
 "candidates": [
  {
   "id": "ttl-lru",
   "title": "Write a TTL and LRU cache class (random operation sequences)",
   "kind": "code",
   "difficulty": "harder",
   "format": "js",
   "validator": "sandbox",
   "checks": 13,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "glob-match",
   "title": "Write a glob matcher (braces, **, character sets, hidden files)",
   "kind": "code",
   "difficulty": "harder",
   "format": "js",
   "validator": "sandbox",
   "checks": 68,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "unified-diff",
   "title": "Write a unified diff (minimal script, fixed tie-break, hunk headers)",
   "kind": "code",
   "difficulty": "harder",
   "format": "js",
   "validator": "sandbox",
   "checks": 20,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "int-expr",
   "title": "Evaluate Python-style integer expressions (precedence, chained comparisons)",
   "kind": "code",
   "difficulty": "harder",
   "format": "js",
   "validator": "sandbox",
   "checks": 64,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "sessions-sql",
   "title": "SQLite sessions report (gaps and islands, median, logout rule)",
   "kind": "sql",
   "difficulty": "harder",
   "format": "sql",
   "validator": "sandbox",
   "checks": 7,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "fx-sql",
   "title": "SQLite as-of price and currency report (missing days, rounding)",
   "kind": "sql",
   "difficulty": "harder",
   "format": "sql",
   "validator": "sandbox",
   "checks": 14,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "skyscrapers",
   "title": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "kind": "reasoning",
   "difficulty": "harder",
   "format": "grid",
   "validator": "exact",
   "checks": 1,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 0,
   "pilotFormatMisses": 2,
   "selected": true,
   "reason": "Sonnet 5.5 passed 0 of 2 in the pilot"
  },
  {
   "id": "two-machines",
   "title": "Pick the most profitable jobs for two machines (one optimum)",
   "kind": "reasoning",
   "difficulty": "harder",
   "format": "line",
   "validator": "exact",
   "checks": 1,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "utf8-decode",
   "title": "Decode UTF-8 with one U+FFFD per maximal subpart",
   "kind": "spec",
   "difficulty": "harder",
   "format": "js",
   "validator": "sandbox",
   "checks": 35,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "cron-next",
   "title": "Next run of a cron expression in UTC (day-of-month or day-of-week rule)",
   "kind": "spec",
   "difficulty": "harder",
   "format": "js",
   "validator": "sandbox",
   "checks": 37,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "queue-sim",
   "title": "Simulate a retry queue (priorities, timeouts, backoff, ties)",
   "kind": "simulation",
   "difficulty": "harder",
   "format": "line",
   "validator": "exact",
   "checks": 1,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "sum-precise",
   "title": "Correctly rounded sum of doubles (ties, overflow, negative zero)",
   "kind": "numeric",
   "difficulty": "harder",
   "format": "js",
   "validator": "sandbox",
   "checks": 32,
   "round": 1,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "nonogram",
   "title": "Solve a 10x10 nonogram (one solution)",
   "kind": "reasoning",
   "difficulty": "harder",
   "format": "nonogram",
   "validator": "exact",
   "checks": 1,
   "round": 2,
   "pilotCalls": 2,
   "pilotStrictPasses": 1,
   "pilotFormatMisses": 0,
   "selected": true,
   "reason": "Sonnet 5.5 passed 1 of 2 in the pilot"
  },
  {
   "id": "three-machines",
   "title": "Pick the most profitable jobs for three machines (30 jobs, one optimum)",
   "kind": "reasoning",
   "difficulty": "harder",
   "format": "line",
   "validator": "exact",
   "checks": 1,
   "round": 2,
   "pilotCalls": 2,
   "pilotStrictPasses": 2,
   "pilotFormatMisses": 0,
   "selected": false,
   "reason": "Sonnet 5.5 passed 2 of 2 in the pilot"
  },
  {
   "id": "lcg-shuffle",
   "title": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "kind": "code-reading",
   "difficulty": "harder",
   "format": "line",
   "validator": "exact",
   "checks": 1,
   "round": 2,
   "pilotCalls": 2,
   "pilotStrictPasses": 0,
   "pilotFormatMisses": 0,
   "selected": true,
   "reason": "Sonnet 5.5 passed 0 of 2 in the pilot"
  },
  {
   "id": "sudoku",
   "title": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "kind": "reasoning",
   "difficulty": "harder",
   "format": "sudoku",
   "validator": "exact",
   "checks": 1,
   "round": 2,
   "pilotCalls": 2,
   "pilotStrictPasses": 1,
   "pilotFormatMisses": 0,
   "selected": true,
   "reason": "Sonnet 5.5 passed 1 of 2 in the pilot"
  }
 ],
 "controls": {
  "ranAt": "2026-10-07T00:34:18.057Z",
  "allOk": true,
  "firstTwelveRanAt": "2026-10-06T21:57:31.146Z",
  "cases": [
   {
    "caseId": "ttl-lru",
    "referencePassed": true,
    "referenceChecks": 13,
    "wrongAnswers": 5,
    "wrongAnswersFailed": 5,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "glob-match",
    "referencePassed": true,
    "referenceChecks": 68,
    "wrongAnswers": 4,
    "wrongAnswersFailed": 4,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "unified-diff",
    "referencePassed": true,
    "referenceChecks": 20,
    "wrongAnswers": 4,
    "wrongAnswersFailed": 4,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "int-expr",
    "referencePassed": true,
    "referenceChecks": 64,
    "wrongAnswers": 5,
    "wrongAnswersFailed": 5,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "sessions-sql",
    "referencePassed": true,
    "referenceChecks": 7,
    "wrongAnswers": 5,
    "wrongAnswersFailed": 5,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "fx-sql",
    "referencePassed": true,
    "referenceChecks": 14,
    "wrongAnswers": 5,
    "wrongAnswersFailed": 5,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "skyscrapers",
    "referencePassed": true,
    "referenceChecks": 1,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "two-machines",
    "referencePassed": true,
    "referenceChecks": 1,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "utf8-decode",
    "referencePassed": true,
    "referenceChecks": 35,
    "wrongAnswers": 4,
    "wrongAnswersFailed": 4,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "cron-next",
    "referencePassed": true,
    "referenceChecks": 37,
    "wrongAnswers": 4,
    "wrongAnswersFailed": 4,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "queue-sim",
    "referencePassed": true,
    "referenceChecks": 1,
    "wrongAnswers": 4,
    "wrongAnswersFailed": 4,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "sum-precise",
    "referencePassed": true,
    "referenceChecks": 32,
    "wrongAnswers": 4,
    "wrongAnswersFailed": 4,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "nonogram",
    "referencePassed": true,
    "referenceChecks": 1,
    "wrongAnswers": 4,
    "wrongAnswersFailed": 4,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "three-machines",
    "referencePassed": true,
    "referenceChecks": 1,
    "wrongAnswers": 5,
    "wrongAnswersFailed": 5,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "lcg-shuffle",
    "referencePassed": true,
    "referenceChecks": 1,
    "wrongAnswers": 4,
    "wrongAnswersFailed": 4,
    "wrappedReferenceIsFormatMiss": true
   },
   {
    "caseId": "sudoku",
    "referencePassed": true,
    "referenceChecks": 1,
    "wrongAnswers": 3,
    "wrongAnswersFailed": 3,
    "wrappedReferenceIsFormatMiss": true
   }
  ],
  "initialRanAt": "2026-10-06T21:46:36.401Z",
  "note": "Initial controls covered 12 candidates before the first probe. The firstTwelveRanAt value is the revised controls run after the validator fix. Final controls covered all 16 before the replacement pilot and counted calls."
 },
 "batches": [
  {
   "batch": "batch-1",
   "schema": "agent-harder-h2h@1",
   "route": "claude-cli",
   "phase": "counted",
   "startedAt": "2026-10-07T01:00:54.652Z",
   "finishedAt": "2026-10-07T02:35:25.397Z",
   "stopReason": null,
   "trimmed": [],
   "resumes": [
    {
     "at": "2026-10-07T01:41:31.272Z",
     "afterStop": "not completed (stop-on-blocked): The model's tool call could not be parsed (retry also failed).",
     "afterCall": "2 sudoku opus"
    }
   ]
  },
  {
   "batch": "batch-2",
   "schema": "agent-harder-h2h@1",
   "route": "codex-cli",
   "phase": "counted",
   "startedAt": "2026-10-07T00:52:05.731Z",
   "finishedAt": "2026-10-07T01:36:12.225Z",
   "stopReason": null,
   "trimmed": [],
   "resumes": []
  }
 ],
 "receipts": [
  {
   "batch": "batch-1",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:00:54.653Z",
   "finishedAt": "2026-10-07T01:04:24.916Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 209432,
   "totalMs": 210077,
   "modelTurnMs": 209387,
   "inputTokens": 4572,
   "outputTokens": 27729,
   "cacheReadTokens": 2926,
   "cacheWriteTokens": 1642,
   "reasoningTokens": 27661,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 109,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:04:24.917Z",
   "finishedAt": "2026-10-07T01:05:23.252Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 57478,
   "totalMs": 58120,
   "modelTurnMs": 57474,
   "inputTokens": 2250,
   "outputTokens": 6118,
   "cacheReadTokens": 531,
   "cacheWriteTokens": 1717,
   "reasoningTokens": 6050,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 109,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:05:23.252Z",
   "finishedAt": "2026-10-07T01:10:23.427Z",
   "status": "failed",
   "error": "Timed out after 300 s. Aborted.",
   "firstUsefulMs": null,
   "totalMs": 299983,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": null,
   "outputChars": 0,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 0,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:10:23.428Z",
   "finishedAt": "2026-10-07T01:11:39.018Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2694,
   "totalMs": 75398,
   "modelTurnMs": 74695,
   "inputTokens": 5107,
   "outputTokens": 16053,
   "cacheReadTokens": 3547,
   "cacheWriteTokens": 1554,
   "reasoningTokens": 89,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 86,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 3,
   "replyHasToolMarkup": true
  },
  {
   "batch": "batch-1",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:11:39.019Z",
   "finishedAt": "2026-10-07T01:16:39.196Z",
   "status": "failed",
   "error": "Timed out after 300 s. Aborted.",
   "firstUsefulMs": 2443,
   "totalMs": 299972,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": null,
   "outputChars": 2996,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 84,
   "replyHasToolMarkup": true
  },
  {
   "batch": "batch-1",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:16:39.196Z",
   "finishedAt": "2026-10-07T01:19:35.236Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 175388,
   "totalMs": 175854,
   "modelTurnMs": 175131,
   "inputTokens": 7667,
   "outputTokens": 20818,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 20777,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 89,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:19:35.236Z",
   "finishedAt": "2026-10-07T01:24:35.501Z",
   "status": "failed",
   "error": "Timed out after 300 s. Aborted.",
   "firstUsefulMs": null,
   "totalMs": 300091,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": null,
   "outputChars": 0,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 0,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:24:35.501Z",
   "finishedAt": "2026-10-07T01:26:06.866Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 90541,
   "totalMs": 91195,
   "modelTurnMs": 90466,
   "inputTokens": 2223,
   "outputTokens": 9543,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 758,
   "reasoningTokens": 9524,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 41,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:26:06.866Z",
   "finishedAt": "2026-10-07T01:29:50.982Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 223654,
   "totalMs": 223948,
   "modelTurnMs": 223233,
   "inputTokens": 7899,
   "outputTokens": 26532,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 26510,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 41,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:29:50.982Z",
   "finishedAt": "2026-10-07T01:30:32.210Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 40809,
   "totalMs": 41053,
   "modelTurnMs": 40440,
   "inputTokens": 2085,
   "outputTokens": 6440,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 620,
   "reasoningTokens": 6415,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 26,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:30:32.211Z",
   "finishedAt": "2026-10-07T01:30:36.241Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1729,
   "totalMs": 3822,
   "modelTurnMs": 2930,
   "inputTokens": 2081,
   "outputTokens": 279,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 616,
   "reasoningTokens": 29,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 435,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 7,
   "replyHasToolMarkup": true
  },
  {
   "batch": "batch-1",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T01:30:36.241Z",
   "finishedAt": "2026-10-07T01:31:48.818Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 70544,
   "totalMs": 72385,
   "modelTurnMs": 71758,
   "inputTokens": 3821,
   "outputTokens": 10894,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 10609,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 678,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 24,
   "replyHasToolMarkup": true
  },
  {
   "batch": "batch-1",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T01:32:01.756Z",
   "finishedAt": "2026-10-07T01:33:10.038Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 67495,
   "totalMs": 68093,
   "modelTurnMs": 67397,
   "inputTokens": 2254,
   "outputTokens": 8957,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 789,
   "reasoningTokens": 8889,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 109,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T01:33:10.038Z",
   "finishedAt": "2026-10-07T01:34:44.446Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 93501,
   "totalMs": 94209,
   "modelTurnMs": 93609,
   "inputTokens": 2250,
   "outputTokens": 9965,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 785,
   "reasoningTokens": 9897,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 109,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T01:34:44.446Z",
   "finishedAt": "2026-10-07T01:37:31.080Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 165572,
   "totalMs": 166437,
   "modelTurnMs": 165798,
   "inputTokens": 7969,
   "outputTokens": 18710,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 18639,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 101,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T01:37:31.081Z",
   "finishedAt": "2026-10-07T01:38:48.090Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3308,
   "totalMs": 76778,
   "modelTurnMs": 75561,
   "inputTokens": 15193,
   "outputTokens": 16059,
   "cacheReadTokens": 3547,
   "cacheWriteTokens": 11640,
   "reasoningTokens": 74,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 86,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 3,
   "replyHasToolMarkup": true
  },
  {
   "batch": "batch-1",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T01:38:48.091Z",
   "finishedAt": "2026-10-07T01:40:45.428Z",
   "status": "failed",
   "error": "The model's tool call could not be parsed (retry also failed).",
   "firstUsefulMs": 1760,
   "totalMs": 116913,
   "modelTurnMs": 116181,
   "inputTokens": 29082,
   "outputTokens": 16746,
   "cacheReadTokens": 5623,
   "cacheWriteTokens": 23449,
   "reasoningTokens": 108,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 62,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T01:47:48.168Z",
   "finishedAt": "2026-10-07T01:48:14.098Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 25317,
   "totalMs": 25734,
   "modelTurnMs": 25084,
   "inputTokens": 3814,
   "outputTokens": 2965,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 2924,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 89,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T01:48:14.098Z",
   "finishedAt": "2026-10-07T01:51:41.886Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 207274,
   "totalMs": 207604,
   "modelTurnMs": 206867,
   "inputTokens": 4516,
   "outputTokens": 27921,
   "cacheReadTokens": 2926,
   "cacheWriteTokens": 1586,
   "reasoningTokens": 27902,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 41,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T01:51:41.886Z",
   "finishedAt": "2026-10-07T01:56:21.573Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3584,
   "totalMs": 279502,
   "modelTurnMs": 278787,
   "inputTokens": 47149,
   "outputTokens": 40044,
   "cacheReadTokens": 18126,
   "cacheWriteTokens": 29013,
   "reasoningTokens": 8584,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 41,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T01:56:21.574Z",
   "finishedAt": "2026-10-07T02:01:21.740Z",
   "status": "failed",
   "error": "Timed out after 300 s. Aborted.",
   "firstUsefulMs": null,
   "totalMs": 299990,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": null,
   "outputChars": 0,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 0,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T02:01:21.741Z",
   "finishedAt": "2026-10-07T02:01:26.255Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1900,
   "totalMs": 4323,
   "modelTurnMs": 3483,
   "inputTokens": 2085,
   "outputTokens": 560,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 620,
   "reasoningTokens": 63,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 996,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 12,
   "replyHasToolMarkup": true
  },
  {
   "batch": "batch-1",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T02:01:26.256Z",
   "finishedAt": "2026-10-07T02:01:31.062Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1572,
   "totalMs": 4627,
   "modelTurnMs": 3818,
   "inputTokens": 2081,
   "outputTokens": 444,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 616,
   "reasoningTokens": 21,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 822,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 11,
   "replyHasToolMarkup": true
  },
  {
   "batch": "batch-1",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T02:01:31.063Z",
   "finishedAt": "2026-10-07T02:04:16.791Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 158981,
   "totalMs": 165434,
   "modelTurnMs": 164828,
   "inputTokens": 7683,
   "outputTokens": 20644,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 19789,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1498,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 15,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:04:21.832Z",
   "finishedAt": "2026-10-07T02:05:34.839Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 72126,
   "totalMs": 72770,
   "modelTurnMs": 72043,
   "inputTokens": 2255,
   "outputTokens": 9616,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 790,
   "reasoningTokens": 9548,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 109,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:05:34.840Z",
   "finishedAt": "2026-10-07T02:06:55.474Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 79550,
   "totalMs": 80344,
   "modelTurnMs": 79706,
   "inputTokens": 2251,
   "outputTokens": 8420,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 786,
   "reasoningTokens": 8352,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 109,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:06:55.474Z",
   "finishedAt": "2026-10-07T02:08:09.133Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 72863,
   "totalMs": 73473,
   "modelTurnMs": 72830,
   "inputTokens": 3963,
   "outputTokens": 8425,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 8354,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 109,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:08:09.133Z",
   "finishedAt": "2026-10-07T02:08:15.117Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2578,
   "totalMs": 5795,
   "modelTurnMs": 5013,
   "inputTokens": 2086,
   "outputTokens": 695,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 621,
   "reasoningTokens": 94,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1145,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 41,
   "replyHasToolMarkup": true
  },
  {
   "batch": "batch-1",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:08:15.118Z",
   "finishedAt": "2026-10-07T02:10:42.298Z",
   "status": "failed",
   "error": "The model's tool call could not be parsed (retry also failed).",
   "firstUsefulMs": 3070,
   "totalMs": 146999,
   "modelTurnMs": 146170,
   "inputTokens": 6359,
   "outputTokens": 16859,
   "cacheReadTokens": 4389,
   "cacheWriteTokens": 1964,
   "reasoningTokens": 16120,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 62,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:10:42.299Z",
   "finishedAt": "2026-10-07T02:11:57.522Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 74672,
   "totalMs": 75049,
   "modelTurnMs": 74439,
   "inputTokens": 3814,
   "outputTokens": 8806,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 8765,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 89,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:11:57.522Z",
   "finishedAt": "2026-10-07T02:16:57.674Z",
   "status": "failed",
   "error": "Timed out after 300 s. Aborted.",
   "firstUsefulMs": null,
   "totalMs": 299950,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": null,
   "outputChars": 0,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 0,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:16:57.675Z",
   "finishedAt": "2026-10-07T02:18:40.009Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4434,
   "totalMs": 102141,
   "modelTurnMs": 101407,
   "inputTokens": 2223,
   "outputTokens": 10708,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 758,
   "reasoningTokens": 8988,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": "line",
     "passed": true,
     "checks": 1,
     "failures": []
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 118,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 3,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:18:40.010Z",
   "finishedAt": "2026-10-07T02:20:27.605Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 107071,
   "totalMs": 107403,
   "modelTurnMs": 106724,
   "inputTokens": 3928,
   "outputTokens": 11619,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 11597,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 41,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:20:27.607Z",
   "finishedAt": "2026-10-07T02:21:09.684Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 41518,
   "totalMs": 41770,
   "modelTurnMs": 40996,
   "inputTokens": 2085,
   "outputTokens": 6724,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 620,
   "reasoningTokens": 6699,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 26,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:21:09.685Z",
   "finishedAt": "2026-10-07T02:21:50.062Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1720,
   "totalMs": 40149,
   "modelTurnMs": 39416,
   "inputTokens": 2080,
   "outputTokens": 4377,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 615,
   "reasoningTokens": 3751,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 26,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-07T02:21:50.063Z",
   "finishedAt": "2026-10-07T02:23:40.819Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 110244,
   "totalMs": 110560,
   "modelTurnMs": 109617,
   "inputTokens": 3820,
   "outputTokens": 13396,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 13368,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 26,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 4,
   "startedAt": "2026-10-07T02:23:41.126Z",
   "finishedAt": "2026-10-07T02:25:20.534Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 98551,
   "totalMs": 99197,
   "modelTurnMs": 98420,
   "inputTokens": 2255,
   "outputTokens": 12639,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 790,
   "reasoningTokens": 12571,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 109,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 4,
   "startedAt": "2026-10-07T02:25:20.535Z",
   "finishedAt": "2026-10-07T02:30:20.722Z",
   "status": "failed",
   "error": "Timed out after 300 s. Aborted.",
   "firstUsefulMs": null,
   "totalMs": 300010,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": null,
   "outputChars": 0,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 0,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 4,
   "startedAt": "2026-10-07T02:30:20.723Z",
   "finishedAt": "2026-10-07T02:35:20.858Z",
   "status": "failed",
   "error": "Timed out after 300 s. Aborted.",
   "firstUsefulMs": null,
   "totalMs": 299938,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": null,
   "outputChars": 0,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 0,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-1",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 4,
   "startedAt": "2026-10-07T02:35:20.859Z",
   "finishedAt": "2026-10-07T02:35:25.396Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2183,
   "totalMs": 4362,
   "modelTurnMs": 3515,
   "inputTokens": 2085,
   "outputTokens": 407,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 620,
   "reasoningTokens": 88,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 621,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "replyLineCount": 14,
   "replyHasToolMarkup": true
  },
  {
   "batch": "batch-2",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-07T00:52:05.731Z",
   "finishedAt": "2026-10-07T00:53:56.006Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 109342,
   "totalMs": 110272,
   "modelTurnMs": 109741,
   "inputTokens": 12324,
   "outputTokens": 4490,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 4436,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 109,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-07T00:53:56.006Z",
   "finishedAt": "2026-10-07T00:58:29.471Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 272916,
   "totalMs": 273463,
   "modelTurnMs": 272164,
   "inputTokens": 12158,
   "outputTokens": 13413,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 13372,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 89,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-07T00:58:29.472Z",
   "finishedAt": "2026-10-07T01:01:35.771Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 185255,
   "totalMs": 186298,
   "modelTurnMs": 185804,
   "inputTokens": 12263,
   "outputTokens": 8309,
   "cacheReadTokens": 4864,
   "cacheWriteTokens": 0,
   "reasoningTokens": 8286,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 41,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-07T01:01:35.771Z",
   "finishedAt": "2026-10-07T01:02:48.547Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 72394,
   "totalMs": 72774,
   "modelTurnMs": 72194,
   "inputTokens": 12164,
   "outputTokens": 3364,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 3335,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 26,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-07T01:02:48.830Z",
   "finishedAt": "2026-10-07T01:06:17.817Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 207822,
   "totalMs": 208985,
   "modelTurnMs": 207671,
   "inputTokens": 12322,
   "outputTokens": 8702,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 8648,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 109,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-07T01:06:17.818Z",
   "finishedAt": "2026-10-07T01:11:17.836Z",
   "status": "failed",
   "error": "Timed out after 300 s. Aborted.",
   "firstUsefulMs": null,
   "totalMs": 300017,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 0,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 0,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-07T01:11:17.836Z",
   "finishedAt": "2026-10-07T01:13:18.074Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 120028,
   "totalMs": 120236,
   "modelTurnMs": 119730,
   "inputTokens": 12265,
   "outputTokens": 4994,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 4971,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 41,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-07T01:13:18.075Z",
   "finishedAt": "2026-10-07T01:14:04.319Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 45839,
   "totalMs": 46243,
   "modelTurnMs": 45004,
   "inputTokens": 12166,
   "outputTokens": 2099,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 2070,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 26,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 3,
   "startedAt": "2026-10-07T01:14:09.257Z",
   "finishedAt": "2026-10-07T01:17:02.986Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 172763,
   "totalMs": 173727,
   "modelTurnMs": 172421,
   "inputTokens": 12324,
   "outputTokens": 7822,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 7768,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 109,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 3,
   "startedAt": "2026-10-07T01:17:02.986Z",
   "finishedAt": "2026-10-07T01:22:03.020Z",
   "status": "failed",
   "error": "Timed out after 300 s. Aborted.",
   "firstUsefulMs": null,
   "totalMs": 300033,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 0,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 0,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 3,
   "startedAt": "2026-10-07T01:22:03.020Z",
   "finishedAt": "2026-10-07T01:23:31.138Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 87796,
   "totalMs": 88116,
   "modelTurnMs": 87492,
   "inputTokens": 12263,
   "outputTokens": 4357,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 4334,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 41,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 3,
   "startedAt": "2026-10-07T01:23:31.139Z",
   "finishedAt": "2026-10-07T01:24:41.062Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 69487,
   "totalMs": 69923,
   "modelTurnMs": 68180,
   "inputTokens": 12166,
   "outputTokens": 3362,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 3333,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 26,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 4,
   "startedAt": "2026-10-07T01:24:41.339Z",
   "finishedAt": "2026-10-07T01:26:44.671Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 122470,
   "totalMs": 123330,
   "modelTurnMs": 122644,
   "inputTokens": 12322,
   "outputTokens": 5419,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 5363,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 109,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 4,
   "startedAt": "2026-10-07T01:26:44.672Z",
   "finishedAt": "2026-10-07T01:31:44.699Z",
   "status": "failed",
   "error": "Timed out after 300 s. Aborted.",
   "firstUsefulMs": null,
   "totalMs": 300027,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "validatorDetail": null,
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 0,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 0,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 4,
   "startedAt": "2026-10-07T01:31:44.700Z",
   "finishedAt": "2026-10-07T01:35:23.961Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 218630,
   "totalMs": 219259,
   "modelTurnMs": 218734,
   "inputTokens": 12261,
   "outputTokens": 10147,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 10124,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "validatorDetail": {
    "strict": {
     "passed": true,
     "checks": 1,
     "failures": []
    },
    "lenient": null
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 41,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "batch-2",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 4,
   "startedAt": "2026-10-07T01:35:23.963Z",
   "finishedAt": "2026-10-07T01:36:12.225Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 47851,
   "totalMs": 48261,
   "modelTurnMs": 47028,
   "inputTokens": 12164,
   "outputTokens": 2099,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 2070,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "validatorDetail": {
    "strict": {
     "passed": false,
     "checks": 1,
     "failures": [
      "Output does not exactly match the expected text."
     ]
    },
    "lenient": {
     "extractor": null,
     "passed": false
    }
   },
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 26,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  }
 ],
 "pilot": [
  {
   "batch": "pilot-1",
   "caseId": "ttl-lru",
   "caseTitle": "Write a TTL and LRU cache class (random operation sequences)",
   "caseKind": "code",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:55:38.858Z",
   "finishedAt": "2026-10-06T21:55:46.435Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3533,
   "totalMs": 7136,
   "modelTurnMs": 6406,
   "inputTokens": 2567,
   "outputTokens": 1056,
   "cacheReadTokens": 531,
   "cacheWriteTokens": 2034,
   "reasoningTokens": 301,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 13,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1809,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": false,
   "replyLineCount": 75,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "glob-match",
   "caseTitle": "Write a glob matcher (braces, **, character sets, hidden files)",
   "caseKind": "code",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:55:46.436Z",
   "finishedAt": "2026-10-06T21:56:15.629Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 19908,
   "totalMs": 28636,
   "modelTurnMs": 28001,
   "inputTokens": 2761,
   "outputTokens": 3986,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1296,
   "reasoningTokens": 2083,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 68,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4617,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 152,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "unified-diff",
   "caseTitle": "Write a unified diff (minimal script, fixed tie-break, hunk headers)",
   "caseKind": "code",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:56:15.629Z",
   "finishedAt": "2026-10-06T21:56:26.154Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5082,
   "totalMs": 10054,
   "modelTurnMs": 9300,
   "inputTokens": 2604,
   "outputTokens": 1583,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1139,
   "reasoningTokens": 539,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 20,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 2100,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 68,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "int-expr",
   "caseTitle": "Evaluate Python-style integer expressions (precedence, chained comparisons)",
   "caseKind": "code",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:56:26.155Z",
   "finishedAt": "2026-10-06T21:56:47.061Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10465,
   "totalMs": 20376,
   "modelTurnMs": 19767,
   "inputTokens": 2646,
   "outputTokens": 3341,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1181,
   "reasoningTokens": 1177,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 64,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 5050,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 166,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "sessions-sql",
   "caseTitle": "SQLite sessions report (gaps and islands, median, logout rule)",
   "caseKind": "sql",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:56:47.062Z",
   "finishedAt": "2026-10-06T21:56:59.079Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8428,
   "totalMs": 11580,
   "modelTurnMs": 10559,
   "inputTokens": 2403,
   "outputTokens": 1595,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 938,
   "reasoningTokens": 938,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 7,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1087,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 35,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "fx-sql",
   "caseTitle": "SQLite as-of price and currency report (missing days, rounding)",
   "caseKind": "sql",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:56:59.079Z",
   "finishedAt": "2026-10-06T21:57:09.967Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7437,
   "totalMs": 10468,
   "modelTurnMs": 9661,
   "inputTokens": 2633,
   "outputTokens": 1336,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1168,
   "reasoningTokens": 758,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1016,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 30,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:57:09.967Z",
   "finishedAt": "2026-10-06T21:58:43.225Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 91061,
   "totalMs": 93091,
   "modelTurnMs": 92074,
   "inputTokens": 2228,
   "outputTokens": 12505,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 763,
   "reasoningTokens": 12300,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 327,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": false,
   "replyLineCount": 12,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "two-machines",
   "caseTitle": "Pick the most profitable jobs for two machines (one optimum)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:58:43.225Z",
   "finishedAt": "2026-10-06T21:59:04.709Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 21058,
   "totalMs": 21317,
   "modelTurnMs": 20682,
   "inputTokens": 2288,
   "outputTokens": 2567,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 823,
   "reasoningTokens": 2545,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 25,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "utf8-decode",
   "caseTitle": "Decode UTF-8 with one U+FFFD per maximal subpart",
   "caseKind": "spec",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:59:04.710Z",
   "finishedAt": "2026-10-06T21:59:10.172Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1440,
   "totalMs": 4911,
   "modelTurnMs": 4037,
   "inputTokens": 2635,
   "outputTokens": 680,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1170,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 35,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1281,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 57,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "cron-next",
   "caseTitle": "Next run of a cron expression in UTC (day-of-month or day-of-week rule)",
   "caseKind": "spec",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:59:10.173Z",
   "finishedAt": "2026-10-06T21:59:31.784Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12791,
   "totalMs": 21114,
   "modelTurnMs": 20452,
   "inputTokens": 2547,
   "outputTokens": 2986,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1082,
   "reasoningTokens": 1254,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 37,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3771,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 103,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "queue-sim",
   "caseTitle": "Simulate a retry queue (priorities, timeouts, backoff, ties)",
   "caseKind": "simulation",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:59:31.785Z",
   "finishedAt": "2026-10-06T21:59:44.615Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12005,
   "totalMs": 12664,
   "modelTurnMs": 11961,
   "inputTokens": 2771,
   "outputTokens": 1785,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1306,
   "reasoningTokens": 1688,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 124,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "sum-precise",
   "caseTitle": "Correctly rounded sum of doubles (ties, overflow, negative zero)",
   "caseKind": "numeric",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:59:44.616Z",
   "finishedAt": "2026-10-06T22:00:04.681Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 15214,
   "totalMs": 19502,
   "modelTurnMs": 18821,
   "inputTokens": 2343,
   "outputTokens": 2563,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 878,
   "reasoningTokens": 1677,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 32,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1761,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 56,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "ttl-lru",
   "caseTitle": "Write a TTL and LRU cache class (random operation sequences)",
   "caseKind": "code",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:00:04.681Z",
   "finishedAt": "2026-10-06T22:00:12.061Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3340,
   "totalMs": 6938,
   "modelTurnMs": 5929,
   "inputTokens": 2569,
   "outputTokens": 988,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1104,
   "reasoningTokens": 288,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 13,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1659,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 73,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "glob-match",
   "caseTitle": "Write a glob matcher (braces, **, character sets, hidden files)",
   "caseKind": "code",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:00:12.061Z",
   "finishedAt": "2026-10-06T22:00:40.360Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 19097,
   "totalMs": 27777,
   "modelTurnMs": 27089,
   "inputTokens": 2761,
   "outputTokens": 3967,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1296,
   "reasoningTokens": 2049,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 68,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4576,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 152,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "unified-diff",
   "caseTitle": "Write a unified diff (minimal script, fixed tie-break, hunk headers)",
   "caseKind": "code",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:00:40.360Z",
   "finishedAt": "2026-10-06T22:00:51.670Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5792,
   "totalMs": 10809,
   "modelTurnMs": 9975,
   "inputTokens": 2604,
   "outputTokens": 1681,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1139,
   "reasoningTokens": 637,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 20,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 2109,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 74,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "int-expr",
   "caseTitle": "Evaluate Python-style integer expressions (precedence, chained comparisons)",
   "caseKind": "code",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:00:51.671Z",
   "finishedAt": "2026-10-06T22:01:09.142Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8366,
   "totalMs": 16962,
   "modelTurnMs": 16326,
   "inputTokens": 2645,
   "outputTokens": 2798,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1180,
   "reasoningTokens": 916,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 64,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4198,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 151,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "sessions-sql",
   "caseTitle": "SQLite sessions report (gaps and islands, median, logout rule)",
   "caseKind": "sql",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:01:09.142Z",
   "finishedAt": "2026-10-06T22:01:21.096Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8523,
   "totalMs": 11532,
   "modelTurnMs": 10854,
   "inputTokens": 2402,
   "outputTokens": 1697,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 937,
   "reasoningTokens": 1045,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 7,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1083,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 34,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "fx-sql",
   "caseTitle": "SQLite as-of price and currency report (missing days, rounding)",
   "caseKind": "sql",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:01:21.096Z",
   "finishedAt": "2026-10-06T22:01:30.882Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6835,
   "totalMs": 9351,
   "modelTurnMs": 8711,
   "inputTokens": 2632,
   "outputTokens": 1290,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1167,
   "reasoningTokens": 787,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 14,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 896,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 29,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "skyscrapers",
   "caseTitle": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:01:30.883Z",
   "finishedAt": "2026-10-06T22:03:34.453Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 121776,
   "totalMs": 123403,
   "modelTurnMs": 122731,
   "inputTokens": 2226,
   "outputTokens": 15507,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 761,
   "reasoningTokens": 15322,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 304,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": false,
   "replyLineCount": 11,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "two-machines",
   "caseTitle": "Pick the most profitable jobs for two machines (one optimum)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:03:34.454Z",
   "finishedAt": "2026-10-06T22:03:54.692Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 19820,
   "totalMs": 20066,
   "modelTurnMs": 19439,
   "inputTokens": 2288,
   "outputTokens": 2408,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 823,
   "reasoningTokens": 2386,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 25,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "utf8-decode",
   "caseTitle": "Decode UTF-8 with one U+FFFD per maximal subpart",
   "caseKind": "spec",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:03:54.693Z",
   "finishedAt": "2026-10-06T22:04:00.835Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2063,
   "totalMs": 5550,
   "modelTurnMs": 4733,
   "inputTokens": 2636,
   "outputTokens": 676,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1171,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 35,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1263,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 54,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "cron-next",
   "caseTitle": "Next run of a cron expression in UTC (day-of-month or day-of-week rule)",
   "caseKind": "spec",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:04:00.835Z",
   "finishedAt": "2026-10-06T22:04:23.936Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 15698,
   "totalMs": 22597,
   "modelTurnMs": 21903,
   "inputTokens": 2547,
   "outputTokens": 3107,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1082,
   "reasoningTokens": 1613,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 37,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3193,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 78,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "queue-sim",
   "caseTitle": "Simulate a retry queue (priorities, timeouts, backoff, ties)",
   "caseKind": "simulation",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:04:23.936Z",
   "finishedAt": "2026-10-06T22:04:37.909Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12891,
   "totalMs": 13796,
   "modelTurnMs": 12825,
   "inputTokens": 2772,
   "outputTokens": 1964,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 1307,
   "reasoningTokens": 1867,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 124,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-1",
   "caseId": "sum-precise",
   "caseTitle": "Correctly rounded sum of doubles (ties, overflow, negative zero)",
   "caseKind": "numeric",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T22:04:37.909Z",
   "finishedAt": "2026-10-06T22:04:54.795Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12500,
   "totalMs": 16349,
   "modelTurnMs": 15735,
   "inputTokens": 2343,
   "outputTokens": 2174,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 878,
   "reasoningTokens": 1363,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 32,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 1627,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 1,
   "scoredWith": "validators v2 (re-scored from the stored reply)",
   "v1Passed": true,
   "replyLineCount": 51,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-2",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T00:34:51.717Z",
   "finishedAt": "2026-10-07T00:36:03.424Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 70781,
   "totalMs": 71519,
   "modelTurnMs": 70771,
   "inputTokens": 2255,
   "outputTokens": 9070,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 790,
   "reasoningTokens": 9002,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 109,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 2,
   "scoredWith": "validators v2",
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-2",
   "caseId": "three-machines",
   "caseTitle": "Pick the most profitable jobs for three machines (30 jobs, one optimum)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T00:36:03.425Z",
   "finishedAt": "2026-10-07T00:36:41.063Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 37186,
   "totalMs": 37368,
   "modelTurnMs": 36641,
   "inputTokens": 2444,
   "outputTokens": 4599,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 979,
   "reasoningTokens": 4544,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 65,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 2,
   "scoredWith": "validators v2",
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-2",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T00:36:41.063Z",
   "finishedAt": "2026-10-07T00:37:20.303Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 38764,
   "totalMs": 39031,
   "modelTurnMs": 38368,
   "inputTokens": 2083,
   "outputTokens": 6102,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 618,
   "reasoningTokens": 6077,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 26,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 2,
   "scoredWith": "validators v2",
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-2",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-07T00:37:20.303Z",
   "finishedAt": "2026-10-07T00:39:03.526Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 101990,
   "totalMs": 103022,
   "modelTurnMs": 102274,
   "inputTokens": 2086,
   "outputTokens": 15889,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 621,
   "reasoningTokens": 15852,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 89,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 2,
   "scoredWith": "validators v2",
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-2",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T00:39:03.527Z",
   "finishedAt": "2026-10-07T00:40:39.501Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 95178,
   "totalMs": 95786,
   "modelTurnMs": 95131,
   "inputTokens": 2255,
   "outputTokens": 12029,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 790,
   "reasoningTokens": 11961,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 110,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 2,
   "scoredWith": "validators v2",
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-2",
   "caseId": "three-machines",
   "caseTitle": "Pick the most profitable jobs for three machines (30 jobs, one optimum)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T00:40:39.501Z",
   "finishedAt": "2026-10-07T00:41:28.226Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 48103,
   "totalMs": 48514,
   "modelTurnMs": 47829,
   "inputTokens": 2444,
   "outputTokens": 5972,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 979,
   "reasoningTokens": 5917,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 65,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 2,
   "scoredWith": "validators v2",
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-2",
   "caseId": "lcg-shuffle",
   "caseTitle": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
   "caseKind": "code-reading",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T00:41:28.227Z",
   "finishedAt": "2026-10-07T00:42:08.362Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 39656,
   "totalMs": 39898,
   "modelTurnMs": 38786,
   "inputTokens": 2085,
   "outputTokens": 6192,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 620,
   "reasoningTokens": 6167,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 26,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 2,
   "scoredWith": "validators v2",
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  },
  {
   "batch": "pilot-2",
   "caseId": "sudoku",
   "caseTitle": "Solve a 9x9 Sudoku with 22 givens (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-07T00:42:08.363Z",
   "finishedAt": "2026-10-07T00:47:08.569Z",
   "status": "failed",
   "error": "Timed out after 300 s. Aborted.",
   "firstUsefulMs": 3296,
   "totalMs": 300012,
   "modelTurnMs": null,
   "inputTokens": null,
   "outputTokens": null,
   "cacheReadTokens": null,
   "cacheWriteTokens": null,
   "reasoningTokens": null,
   "passed": null,
   "formatMiss": null,
   "answerCorrect": null,
   "checks": null,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": null,
   "outputChars": 919,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "pilot",
   "pilotRound": 2,
   "scoredWith": "validators v2",
   "replyLineCount": 32,
   "replyHasToolMarkup": true
  }
 ],
 "probes": [
  {
   "batch": "probe-claude-20261006",
   "caseId": "ttl-lru",
   "caseTitle": "Write a TTL and LRU cache class (random operation sequences)",
   "caseKind": "code",
   "caseDifficulty": "harder",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T21:54:45.515Z",
   "finishedAt": "2026-10-06T21:55:25.985Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 36720,
   "totalMs": 40009,
   "modelTurnMs": 39350,
   "inputTokens": 4155,
   "outputTokens": 6526,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 4145,
   "reasoningTokens": 5771,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 13,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 2179,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap setting: 16,000 tokens. Reported totals sometimes exceed it; enforcement of the reported total is not established.",
    "Reasoning was observed but its content is never captured."
   ],
   "phase": "probe",
   "replyLineCount": 86,
   "replyHasToolMarkup": false
  },
  {
   "batch": "probe-codex-20261007",
   "caseId": "nonogram",
   "caseTitle": "Solve a 10x10 nonogram (one solution)",
   "caseKind": "reasoning",
   "caseDifficulty": "harder",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-07T00:48:07.028Z",
   "finishedAt": "2026-10-07T00:52:00.720Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 232729,
   "totalMs": 233690,
   "modelTurnMs": 233054,
   "inputTokens": 12322,
   "outputTokens": 10377,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 10323,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 109,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "phase": "probe",
   "replyLineCount": 1,
   "replyHasToolMarkup": false
  }
 ],
 "callCaps": {
  "claude-cli": 105,
  "codex-cli": 33
 },
 "timingEvidence": {
  "protocolBirthAt": "2026-10-06T21:49:28.380Z",
  "firstProbeStartedAt": "2026-10-06T21:54:45.515Z",
  "frozenFileBirthAt": "2026-10-07T00:47:15.459Z",
  "firstCountedStartedAt": "2026-10-07T00:52:05.731Z",
  "note": "File birth times checked on the run host. The initial protocol predates all calls. The public protocol text is a later summary with amendments."
 },
 "hostCaveat": "Arena servers shared the Mac during part of the run. Host load was not controlled; recorded CLI times do not isolate model speed.",
 "protocolNote": "The run protocol was declared before the first run. Its public summary is the Method section of the study page."
}
