{
 "note": "Consistency receipts copied from public-runs/caching-consistency: the same prompt asked 10 times per model. Answers are published as ordinal ids per cell (A1, A2, ...: distinct normalized answers; R1, R2, ...: distinct raw replies), never as text. Every attempt is kept.",
 "prompts": [
  {
   "id": "digit-sum-count",
   "kind": "exact-numeric",
   "format": "number",
   "validator": "exact",
   "expected": "282"
  },
  {
   "id": "order-json",
   "kind": "json-shape",
   "format": "json",
   "validator": "sandbox"
  },
  {
   "id": "median-fix",
   "kind": "code-fix",
   "format": "js",
   "validator": "sandbox"
  }
 ],
 "controls": {
  "claude": {
   "ranAt": "2026-10-06T14:25:44.110Z",
   "allOk": true,
   "prompts": [
    {
     "promptId": "digit-sum-count",
     "referencePassed": true,
     "referenceChecks": 1,
     "wrongAnswers": 3,
     "wrongAnswersFailed": 3,
     "wrappedReferenceIsFormatMiss": true,
     "normalizerControl": true
    },
    {
     "promptId": "order-json",
     "referencePassed": true,
     "referenceChecks": 9,
     "wrongAnswers": 4,
     "wrongAnswersFailed": 4,
     "wrappedReferenceIsFormatMiss": true,
     "normalizerControl": true
    },
    {
     "promptId": "median-fix",
     "referencePassed": true,
     "referenceChecks": 10,
     "wrongAnswers": 4,
     "wrongAnswersFailed": 4,
     "wrappedReferenceIsFormatMiss": true,
     "normalizerControl": true
    }
   ]
  },
  "codex": {
   "ranAt": "2026-10-06T14:26:50.511Z",
   "allOk": true,
   "prompts": [
    {
     "promptId": "digit-sum-count",
     "referencePassed": true,
     "referenceChecks": 1,
     "wrongAnswers": 3,
     "wrongAnswersFailed": 3,
     "wrappedReferenceIsFormatMiss": true,
     "normalizerControl": true
    },
    {
     "promptId": "order-json",
     "referencePassed": true,
     "referenceChecks": 9,
     "wrongAnswers": 4,
     "wrongAnswersFailed": 4,
     "wrappedReferenceIsFormatMiss": true,
     "normalizerControl": true
    },
    {
     "promptId": "median-fix",
     "referencePassed": true,
     "referenceChecks": 10,
     "wrongAnswers": 4,
     "wrongAnswersFailed": 4,
     "wrappedReferenceIsFormatMiss": true,
     "normalizerControl": true
    }
   ]
  }
 },
 "batches": [
  {
   "batch": "batch-1",
   "schema": "agent-consistency@1",
   "route": "codex-cli",
   "startedAt": "2026-10-06T14:26:54.546Z",
   "finishedAt": "2026-10-06T14:32:23.283Z",
   "stopReason": null,
   "trimmed": 0
  },
  {
   "batch": "batch-2",
   "schema": "agent-consistency@1",
   "route": "claude-cli",
   "startedAt": "2026-10-06T14:37:20.885Z",
   "finishedAt": "2026-10-06T14:43:06.033Z",
   "stopReason": null,
   "trimmed": 0
  }
 ],
 "receipts": [
  {
   "batch": "batch-1",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:26:54.546Z",
   "finishedAt": "2026-10-06T14:27:12.522Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 17514,
   "totalMs": 17973,
   "modelTurnMs": 16442,
   "inputTokens": 12013,
   "outputTokens": 361,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 354,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-1",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:27:12.524Z",
   "finishedAt": "2026-10-06T14:27:20.391Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5209,
   "totalMs": 7618,
   "modelTurnMs": 6918,
   "inputTokens": 12232,
   "outputTokens": 95,
   "cacheReadTokens": 3840,
   "cacheWriteTokens": 0,
   "reasoningTokens": 24,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 180,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 1,
   "startedAt": "2026-10-06T14:27:20.391Z",
   "finishedAt": "2026-10-06T14:27:31.335Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8033,
   "totalMs": 10674,
   "modelTurnMs": 10203,
   "inputTokens": 12094,
   "outputTokens": 262,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 125,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 440,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:27:31.335Z",
   "finishedAt": "2026-10-06T14:27:48.597Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 17116,
   "totalMs": 17260,
   "modelTurnMs": 16492,
   "inputTokens": 12015,
   "outputTokens": 255,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 248,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-1",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:27:48.597Z",
   "finishedAt": "2026-10-06T14:27:56.960Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5267,
   "totalMs": 8113,
   "modelTurnMs": 7362,
   "inputTokens": 12234,
   "outputTokens": 92,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 21,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 180,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 2,
   "startedAt": "2026-10-06T14:27:56.961Z",
   "finishedAt": "2026-10-06T14:28:09.357Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9516,
   "totalMs": 12160,
   "modelTurnMs": 11444,
   "inputTokens": 12098,
   "outputTokens": 237,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 100,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 414,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A2",
   "rawAnswerId": "R2"
  },
  {
   "batch": "batch-1",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 3,
   "startedAt": "2026-10-06T14:28:09.358Z",
   "finishedAt": "2026-10-06T14:28:21.652Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12203,
   "totalMs": 12293,
   "modelTurnMs": 11765,
   "inputTokens": 12013,
   "outputTokens": 239,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 232,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-1",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 3,
   "startedAt": "2026-10-06T14:28:21.653Z",
   "finishedAt": "2026-10-06T14:28:27.773Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4468,
   "totalMs": 5870,
   "modelTurnMs": 5353,
   "inputTokens": 12236,
   "outputTokens": 92,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 21,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 180,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 3,
   "startedAt": "2026-10-06T14:28:27.773Z",
   "finishedAt": "2026-10-06T14:28:42.866Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12079,
   "totalMs": 14846,
   "modelTurnMs": 13933,
   "inputTokens": 12096,
   "outputTokens": 281,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 144,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 436,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A3",
   "rawAnswerId": "R3"
  },
  {
   "batch": "batch-1",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 4,
   "startedAt": "2026-10-06T14:28:42.866Z",
   "finishedAt": "2026-10-06T14:28:56.717Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13705,
   "totalMs": 13850,
   "modelTurnMs": 13132,
   "inputTokens": 12011,
   "outputTokens": 253,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 246,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-1",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 4,
   "startedAt": "2026-10-06T14:28:56.718Z",
   "finishedAt": "2026-10-06T14:29:02.467Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4058,
   "totalMs": 5507,
   "modelTurnMs": 4993,
   "inputTokens": 12230,
   "outputTokens": 92,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 21,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 180,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 4,
   "startedAt": "2026-10-06T14:29:02.468Z",
   "finishedAt": "2026-10-06T14:29:14.646Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9077,
   "totalMs": 11896,
   "modelTurnMs": 11425,
   "inputTokens": 12098,
   "outputTokens": 281,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 144,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 440,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 5,
   "startedAt": "2026-10-06T14:29:14.646Z",
   "finishedAt": "2026-10-06T14:29:28.035Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13350,
   "totalMs": 13387,
   "modelTurnMs": 12840,
   "inputTokens": 12013,
   "outputTokens": 284,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 277,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-1",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 5,
   "startedAt": "2026-10-06T14:29:28.036Z",
   "finishedAt": "2026-10-06T14:29:35.187Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5682,
   "totalMs": 6890,
   "modelTurnMs": 6346,
   "inputTokens": 12232,
   "outputTokens": 95,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 24,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 180,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 5,
   "startedAt": "2026-10-06T14:29:35.187Z",
   "finishedAt": "2026-10-06T14:29:45.968Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8008,
   "totalMs": 10537,
   "modelTurnMs": 9975,
   "inputTokens": 12094,
   "outputTokens": 217,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 80,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 414,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A2",
   "rawAnswerId": "R2"
  },
  {
   "batch": "batch-1",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 6,
   "startedAt": "2026-10-06T14:29:45.969Z",
   "finishedAt": "2026-10-06T14:30:00.239Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13952,
   "totalMs": 14269,
   "modelTurnMs": 13551,
   "inputTokens": 12013,
   "outputTokens": 271,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 264,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-1",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 6,
   "startedAt": "2026-10-06T14:30:00.240Z",
   "finishedAt": "2026-10-06T14:30:06.446Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4742,
   "totalMs": 5953,
   "modelTurnMs": 5421,
   "inputTokens": 12234,
   "outputTokens": 95,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 24,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 180,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 6,
   "startedAt": "2026-10-06T14:30:06.446Z",
   "finishedAt": "2026-10-06T14:30:17.235Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7853,
   "totalMs": 10556,
   "modelTurnMs": 9501,
   "inputTokens": 12096,
   "outputTokens": 228,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 91,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 414,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A2",
   "rawAnswerId": "R2"
  },
  {
   "batch": "batch-1",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 7,
   "startedAt": "2026-10-06T14:30:17.235Z",
   "finishedAt": "2026-10-06T14:30:30.562Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13240,
   "totalMs": 13325,
   "modelTurnMs": 12011,
   "inputTokens": 12011,
   "outputTokens": 253,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 246,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-1",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 7,
   "startedAt": "2026-10-06T14:30:30.564Z",
   "finishedAt": "2026-10-06T14:30:39.067Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6685,
   "totalMs": 8255,
   "modelTurnMs": 6684,
   "inputTokens": 12234,
   "outputTokens": 95,
   "cacheReadTokens": 3840,
   "cacheWriteTokens": 0,
   "reasoningTokens": 24,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 180,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 7,
   "startedAt": "2026-10-06T14:30:39.067Z",
   "finishedAt": "2026-10-06T14:30:49.750Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7927,
   "totalMs": 10445,
   "modelTurnMs": 9719,
   "inputTokens": 12096,
   "outputTokens": 228,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 91,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 414,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A2",
   "rawAnswerId": "R2"
  },
  {
   "batch": "batch-1",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 8,
   "startedAt": "2026-10-06T14:30:49.750Z",
   "finishedAt": "2026-10-06T14:31:02.038Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12064,
   "totalMs": 12286,
   "modelTurnMs": 11738,
   "inputTokens": 12013,
   "outputTokens": 242,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 235,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-1",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 8,
   "startedAt": "2026-10-06T14:31:02.040Z",
   "finishedAt": "2026-10-06T14:31:09.377Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5762,
   "totalMs": 7085,
   "modelTurnMs": 5714,
   "inputTokens": 12236,
   "outputTokens": 92,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 21,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 180,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 8,
   "startedAt": "2026-10-06T14:31:09.377Z",
   "finishedAt": "2026-10-06T14:31:18.702Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6550,
   "totalMs": 9078,
   "modelTurnMs": 8582,
   "inputTokens": 12098,
   "outputTokens": 217,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 80,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 416,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A4",
   "rawAnswerId": "R4"
  },
  {
   "batch": "batch-1",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 9,
   "startedAt": "2026-10-06T14:31:18.702Z",
   "finishedAt": "2026-10-06T14:31:31.103Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 12261,
   "totalMs": 12400,
   "modelTurnMs": 11792,
   "inputTokens": 12015,
   "outputTokens": 272,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 265,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-1",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 9,
   "startedAt": "2026-10-06T14:31:31.104Z",
   "finishedAt": "2026-10-06T14:31:36.607Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4035,
   "totalMs": 5252,
   "modelTurnMs": 4664,
   "inputTokens": 12232,
   "outputTokens": 93,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 22,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 180,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 9,
   "startedAt": "2026-10-06T14:31:36.608Z",
   "finishedAt": "2026-10-06T14:31:49.857Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10090,
   "totalMs": 13008,
   "modelTurnMs": 12524,
   "inputTokens": 12098,
   "outputTokens": 281,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 144,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 416,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A5",
   "rawAnswerId": "R5"
  },
  {
   "batch": "batch-1",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 10,
   "startedAt": "2026-10-06T14:31:49.857Z",
   "finishedAt": "2026-10-06T14:32:03.238Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13298,
   "totalMs": 13380,
   "modelTurnMs": 12873,
   "inputTokens": 12013,
   "outputTokens": 274,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 267,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 3,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-1",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 10,
   "startedAt": "2026-10-06T14:32:03.238Z",
   "finishedAt": "2026-10-06T14:32:09.106Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4357,
   "totalMs": 5627,
   "modelTurnMs": 5135,
   "inputTokens": 12234,
   "outputTokens": 92,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 21,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 180,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-1",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "effortReported": "medium",
   "rep": 10,
   "startedAt": "2026-10-06T14:32:09.106Z",
   "finishedAt": "2026-10-06T14:32:23.283Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11151,
   "totalMs": 13894,
   "modelTurnMs": 13339,
   "inputTokens": 12094,
   "outputTokens": 335,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 198,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 439,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ],
   "answerId": "A6",
   "rawAnswerId": "R6"
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T14:37:20.885Z",
   "finishedAt": "2026-10-06T14:37:26.843Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4606,
   "totalMs": 5753,
   "modelTurnMs": 4803,
   "inputTokens": 3660,
   "outputTokens": 631,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 524,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 268,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 289,
   "failedOutputLineCount": 11
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T14:37:26.843Z",
   "finishedAt": "2026-10-06T14:37:34.692Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7328,
   "totalMs": 7635,
   "modelTurnMs": 6848,
   "inputTokens": 1916,
   "outputTokens": 834,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 451,
   "reasoningTokens": 831,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T14:37:34.693Z",
   "finishedAt": "2026-10-06T14:37:43.792Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7343,
   "totalMs": 8435,
   "modelTurnMs": 7130,
   "inputTokens": 3915,
   "outputTokens": 877,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 739,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 0,
   "failures": [
    "parse: Unexpected token '`', \"```json\n{\n\"... is not valid JSON"
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 284,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "failedOutputLineCount": 20
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T14:37:43.792Z",
   "finishedAt": "2026-10-06T14:37:49.533Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4254,
   "totalMs": 5303,
   "modelTurnMs": 4407,
   "inputTokens": 2244,
   "outputTokens": 188,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 779,
   "reasoningTokens": 88,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 180,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T14:37:49.533Z",
   "finishedAt": "2026-10-06T14:37:55.565Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4413,
   "totalMs": 5601,
   "modelTurnMs": 4683,
   "inputTokens": 3759,
   "outputTokens": 544,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 412,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 315,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-06T14:37:55.566Z",
   "finishedAt": "2026-10-06T14:37:58.457Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1076,
   "totalMs": 2460,
   "modelTurnMs": 1613,
   "inputTokens": 2049,
   "outputTokens": 154,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 584,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 325,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T14:37:58.458Z",
   "finishedAt": "2026-10-06T14:38:04.328Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4197,
   "totalMs": 5674,
   "modelTurnMs": 4779,
   "inputTokens": 3660,
   "outputTokens": 612,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 451,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 359,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R2",
   "answerNumber": 289,
   "failedOutputLineCount": 7
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T14:38:04.329Z",
   "finishedAt": "2026-10-06T14:38:10.863Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6035,
   "totalMs": 6318,
   "modelTurnMs": 5570,
   "inputTokens": 1917,
   "outputTokens": 682,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 452,
   "reasoningTokens": 679,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T14:38:10.863Z",
   "finishedAt": "2026-10-06T14:38:21.096Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8482,
   "totalMs": 9544,
   "modelTurnMs": 8863,
   "inputTokens": 3914,
   "outputTokens": 1113,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 975,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 0,
   "failures": [
    "parse: Unexpected token '`', \"```json\n{\n\"... is not valid JSON"
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 284,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "failedOutputLineCount": 20
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T14:38:21.096Z",
   "finishedAt": "2026-10-06T14:38:24.243Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1749,
   "totalMs": 2700,
   "modelTurnMs": 1817,
   "inputTokens": 2243,
   "outputTokens": 155,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 778,
   "reasoningTokens": 55,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 180,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T14:38:24.243Z",
   "finishedAt": "2026-10-06T14:38:31.384Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5617,
   "totalMs": 6676,
   "modelTurnMs": 5957,
   "inputTokens": 3759,
   "outputTokens": 693,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 560,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 330,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A2",
   "rawAnswerId": "R2"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-06T14:38:31.385Z",
   "finishedAt": "2026-10-06T14:38:34.148Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1104,
   "totalMs": 2318,
   "modelTurnMs": 1480,
   "inputTokens": 2049,
   "outputTokens": 154,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 584,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 325,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T14:38:34.148Z",
   "finishedAt": "2026-10-06T14:38:38.998Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3368,
   "totalMs": 4657,
   "modelTurnMs": 3827,
   "inputTokens": 3661,
   "outputTokens": 466,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 326,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 301,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R3",
   "answerNumber": 289,
   "failedOutputLineCount": 6
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T14:38:38.998Z",
   "finishedAt": "2026-10-06T14:38:46.142Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6661,
   "totalMs": 6930,
   "modelTurnMs": 6205,
   "inputTokens": 1916,
   "outputTokens": 754,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 451,
   "reasoningTokens": 751,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T14:38:46.142Z",
   "finishedAt": "2026-10-06T14:38:52.954Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5153,
   "totalMs": 6083,
   "modelTurnMs": 5250,
   "inputTokens": 3914,
   "outputTokens": 799,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 660,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 0,
   "failures": [
    "parse: Unexpected token '`', \"```json\n{\n\"... is not valid JSON"
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 284,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "failedOutputLineCount": 20
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T14:38:52.954Z",
   "finishedAt": "2026-10-06T14:38:56.078Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1834,
   "totalMs": 2683,
   "modelTurnMs": 1898,
   "inputTokens": 2241,
   "outputTokens": 205,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 776,
   "reasoningTokens": 105,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 180,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T14:38:56.079Z",
   "finishedAt": "2026-10-06T14:39:02.165Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4500,
   "totalMs": 5642,
   "modelTurnMs": 4851,
   "inputTokens": 3759,
   "outputTokens": 566,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 434,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 315,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-06T14:39:02.166Z",
   "finishedAt": "2026-10-06T14:39:06.519Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2413,
   "totalMs": 3906,
   "modelTurnMs": 2883,
   "inputTokens": 2048,
   "outputTokens": 146,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 583,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 303,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "answerId": "A2",
   "rawAnswerId": "R2"
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 4,
   "startedAt": "2026-10-06T14:39:06.519Z",
   "finishedAt": "2026-10-06T14:39:11.307Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3336,
   "totalMs": 4594,
   "modelTurnMs": 3751,
   "inputTokens": 3661,
   "outputTokens": 510,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 326,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 450,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R4",
   "answerNumber": 289,
   "failedOutputLineCount": 21
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 4,
   "startedAt": "2026-10-06T14:39:11.308Z",
   "finishedAt": "2026-10-06T14:39:17.330Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5543,
   "totalMs": 5808,
   "modelTurnMs": 5067,
   "inputTokens": 1918,
   "outputTokens": 581,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 453,
   "reasoningTokens": 578,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 4,
   "startedAt": "2026-10-06T14:39:17.331Z",
   "finishedAt": "2026-10-06T14:39:23.372Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4341,
   "totalMs": 5277,
   "modelTurnMs": 4450,
   "inputTokens": 3914,
   "outputTokens": 689,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 550,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 0,
   "failures": [
    "parse: Unexpected token '`', \"```json\n{\n\"... is not valid JSON"
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 284,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "failedOutputLineCount": 20
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 4,
   "startedAt": "2026-10-06T14:39:23.372Z",
   "finishedAt": "2026-10-06T14:39:26.755Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1859,
   "totalMs": 2923,
   "modelTurnMs": 1942,
   "inputTokens": 2243,
   "outputTokens": 202,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 778,
   "reasoningTokens": 102,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 180,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 4,
   "startedAt": "2026-10-06T14:39:26.755Z",
   "finishedAt": "2026-10-06T14:39:32.654Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4517,
   "totalMs": 5455,
   "modelTurnMs": 4571,
   "inputTokens": 3759,
   "outputTokens": 626,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 494,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 315,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 4,
   "startedAt": "2026-10-06T14:39:32.654Z",
   "finishedAt": "2026-10-06T14:39:35.799Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1356,
   "totalMs": 2708,
   "modelTurnMs": 1818,
   "inputTokens": 2048,
   "outputTokens": 154,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 583,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 325,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 5,
   "startedAt": "2026-10-06T14:39:35.799Z",
   "finishedAt": "2026-10-06T14:39:41.301Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3794,
   "totalMs": 5313,
   "modelTurnMs": 4360,
   "inputTokens": 3659,
   "outputTokens": 521,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 358,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 386,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R5",
   "answerNumber": 289,
   "failedOutputLineCount": 6
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 5,
   "startedAt": "2026-10-06T14:39:41.302Z",
   "finishedAt": "2026-10-06T14:39:49.021Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7221,
   "totalMs": 7509,
   "modelTurnMs": 6794,
   "inputTokens": 1917,
   "outputTokens": 754,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 452,
   "reasoningTokens": 751,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 5,
   "startedAt": "2026-10-06T14:39:49.021Z",
   "finishedAt": "2026-10-06T14:39:55.333Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4633,
   "totalMs": 5591,
   "modelTurnMs": 4682,
   "inputTokens": 3914,
   "outputTokens": 794,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 655,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 0,
   "failures": [
    "parse: Unexpected token '`', \"```json\n{\n\"... is not valid JSON"
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 284,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "failedOutputLineCount": 20
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 5,
   "startedAt": "2026-10-06T14:39:55.333Z",
   "finishedAt": "2026-10-06T14:39:58.714Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2121,
   "totalMs": 2934,
   "modelTurnMs": 1969,
   "inputTokens": 2241,
   "outputTokens": 211,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 776,
   "reasoningTokens": 111,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 180,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 5,
   "startedAt": "2026-10-06T14:39:58.714Z",
   "finishedAt": "2026-10-06T14:40:05.939Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5711,
   "totalMs": 6774,
   "modelTurnMs": 5907,
   "inputTokens": 3759,
   "outputTokens": 664,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 533,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 312,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A3",
   "rawAnswerId": "R3"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 5,
   "startedAt": "2026-10-06T14:40:05.939Z",
   "finishedAt": "2026-10-06T14:40:10.530Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2821,
   "totalMs": 4164,
   "modelTurnMs": 3347,
   "inputTokens": 2048,
   "outputTokens": 154,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 583,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 325,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 6,
   "startedAt": "2026-10-06T14:40:10.530Z",
   "finishedAt": "2026-10-06T14:40:15.528Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3703,
   "totalMs": 4809,
   "modelTurnMs": 3931,
   "inputTokens": 3661,
   "outputTokens": 455,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 361,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 238,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R6",
   "answerNumber": 289,
   "failedOutputLineCount": 10
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 6,
   "startedAt": "2026-10-06T14:40:15.529Z",
   "finishedAt": "2026-10-06T14:40:22.134Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6098,
   "totalMs": 6395,
   "modelTurnMs": 5705,
   "inputTokens": 1916,
   "outputTokens": 622,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 451,
   "reasoningTokens": 619,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 6,
   "startedAt": "2026-10-06T14:40:22.134Z",
   "finishedAt": "2026-10-06T14:40:29.591Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6279,
   "totalMs": 7001,
   "modelTurnMs": 6286,
   "inputTokens": 3915,
   "outputTokens": 776,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 695,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 180,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R2"
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 6,
   "startedAt": "2026-10-06T14:40:29.591Z",
   "finishedAt": "2026-10-06T14:40:32.870Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1892,
   "totalMs": 2823,
   "modelTurnMs": 1944,
   "inputTokens": 2242,
   "outputTokens": 205,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 777,
   "reasoningTokens": 105,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 180,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 6,
   "startedAt": "2026-10-06T14:40:32.870Z",
   "finishedAt": "2026-10-06T14:40:40.631Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6325,
   "totalMs": 7325,
   "modelTurnMs": 6626,
   "inputTokens": 3759,
   "outputTokens": 774,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 650,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 289,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A4",
   "rawAnswerId": "R4"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 6,
   "startedAt": "2026-10-06T14:40:40.631Z",
   "finishedAt": "2026-10-06T14:40:43.519Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1045,
   "totalMs": 2467,
   "modelTurnMs": 1462,
   "inputTokens": 2049,
   "outputTokens": 154,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 584,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 325,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 7,
   "startedAt": "2026-10-06T14:40:43.519Z",
   "finishedAt": "2026-10-06T14:40:48.289Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3351,
   "totalMs": 4581,
   "modelTurnMs": 3673,
   "inputTokens": 3661,
   "outputTokens": 435,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 330,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 283,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R7",
   "answerNumber": 289,
   "failedOutputLineCount": 10
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 7,
   "startedAt": "2026-10-06T14:40:48.289Z",
   "finishedAt": "2026-10-06T14:40:55.346Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6514,
   "totalMs": 6850,
   "modelTurnMs": 6042,
   "inputTokens": 1916,
   "outputTokens": 670,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 451,
   "reasoningTokens": 667,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 7,
   "startedAt": "2026-10-06T14:40:55.347Z",
   "finishedAt": "2026-10-06T14:41:04.247Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7105,
   "totalMs": 8217,
   "modelTurnMs": 7434,
   "inputTokens": 3915,
   "outputTokens": 959,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 820,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 0,
   "failures": [
    "parse: Unexpected token '`', \"```json\n{\n\"... is not valid JSON"
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 284,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "failedOutputLineCount": 20
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 7,
   "startedAt": "2026-10-06T14:41:04.247Z",
   "finishedAt": "2026-10-06T14:41:07.520Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1897,
   "totalMs": 2854,
   "modelTurnMs": 1971,
   "inputTokens": 2242,
   "outputTokens": 202,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 777,
   "reasoningTokens": 102,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 180,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 7,
   "startedAt": "2026-10-06T14:41:07.521Z",
   "finishedAt": "2026-10-06T14:41:14.266Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5228,
   "totalMs": 6314,
   "modelTurnMs": 5549,
   "inputTokens": 3758,
   "outputTokens": 685,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 553,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 315,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 7,
   "startedAt": "2026-10-06T14:41:14.266Z",
   "finishedAt": "2026-10-06T14:41:17.327Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1322,
   "totalMs": 2637,
   "modelTurnMs": 1592,
   "inputTokens": 2048,
   "outputTokens": 150,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 583,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 315,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "answerId": "A3",
   "rawAnswerId": "R3"
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 8,
   "startedAt": "2026-10-06T14:41:17.328Z",
   "finishedAt": "2026-10-06T14:41:21.932Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3371,
   "totalMs": 4417,
   "modelTurnMs": 3583,
   "inputTokens": 3660,
   "outputTokens": 425,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 333,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 229,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R8",
   "answerNumber": 289,
   "failedOutputLineCount": 11
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 8,
   "startedAt": "2026-10-06T14:41:21.932Z",
   "finishedAt": "2026-10-06T14:41:29.401Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6965,
   "totalMs": 7263,
   "modelTurnMs": 6544,
   "inputTokens": 1918,
   "outputTokens": 759,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 453,
   "reasoningTokens": 756,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 8,
   "startedAt": "2026-10-06T14:41:29.402Z",
   "finishedAt": "2026-10-06T14:41:42.362Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11364,
   "totalMs": 12274,
   "modelTurnMs": 11490,
   "inputTokens": 3915,
   "outputTokens": 857,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 719,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 0,
   "failures": [
    "parse: Unexpected token '`', \"```json\n{\n\"... is not valid JSON"
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 284,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "failedOutputLineCount": 20
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 8,
   "startedAt": "2026-10-06T14:41:42.363Z",
   "finishedAt": "2026-10-06T14:41:47.484Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3617,
   "totalMs": 4659,
   "modelTurnMs": 3789,
   "inputTokens": 2242,
   "outputTokens": 202,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 777,
   "reasoningTokens": 102,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 180,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 8,
   "startedAt": "2026-10-06T14:41:47.485Z",
   "finishedAt": "2026-10-06T14:41:53.233Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4356,
   "totalMs": 5319,
   "modelTurnMs": 4423,
   "inputTokens": 3758,
   "outputTokens": 620,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 488,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 315,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 8,
   "startedAt": "2026-10-06T14:41:53.233Z",
   "finishedAt": "2026-10-06T14:41:56.518Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1208,
   "totalMs": 2845,
   "modelTurnMs": 1610,
   "inputTokens": 2049,
   "outputTokens": 154,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 584,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 325,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 9,
   "startedAt": "2026-10-06T14:41:56.518Z",
   "finishedAt": "2026-10-06T14:42:02.785Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3446,
   "totalMs": 6072,
   "modelTurnMs": 5379,
   "inputTokens": 3661,
   "outputTokens": 678,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 332,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 784,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R9",
   "answerNumber": 289,
   "failedOutputLineCount": 37
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 9,
   "startedAt": "2026-10-06T14:42:02.785Z",
   "finishedAt": "2026-10-06T14:42:10.803Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7529,
   "totalMs": 7808,
   "modelTurnMs": 7065,
   "inputTokens": 1917,
   "outputTokens": 673,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 452,
   "reasoningTokens": 670,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 9,
   "startedAt": "2026-10-06T14:42:10.803Z",
   "finishedAt": "2026-10-06T14:42:17.953Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5320,
   "totalMs": 6430,
   "modelTurnMs": 5710,
   "inputTokens": 3913,
   "outputTokens": 751,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 612,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 0,
   "failures": [
    "parse: Unexpected token '`', \"```json\n{\n\"... is not valid JSON"
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 284,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "failedOutputLineCount": 20
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 9,
   "startedAt": "2026-10-06T14:42:17.954Z",
   "finishedAt": "2026-10-06T14:42:21.872Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2455,
   "totalMs": 3453,
   "modelTurnMs": 2483,
   "inputTokens": 2240,
   "outputTokens": 169,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 775,
   "reasoningTokens": 69,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 180,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 9,
   "startedAt": "2026-10-06T14:42:21.873Z",
   "finishedAt": "2026-10-06T14:42:28.595Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5191,
   "totalMs": 6249,
   "modelTurnMs": 5497,
   "inputTokens": 3759,
   "outputTokens": 638,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 508,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 330,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A5",
   "rawAnswerId": "R5"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 9,
   "startedAt": "2026-10-06T14:42:28.596Z",
   "finishedAt": "2026-10-06T14:42:31.523Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1236,
   "totalMs": 2482,
   "modelTurnMs": 1626,
   "inputTokens": 2049,
   "outputTokens": 154,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 584,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 325,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 10,
   "startedAt": "2026-10-06T14:42:31.524Z",
   "finishedAt": "2026-10-06T14:42:37.920Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4177,
   "totalMs": 6202,
   "modelTurnMs": 5500,
   "inputTokens": 3659,
   "outputTokens": 706,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 436,
   "passed": false,
   "formatMiss": false,
   "answerCorrect": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 594,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R10",
   "answerNumber": 289,
   "failedOutputLineCount": 27
  },
  {
   "batch": "batch-2",
   "promptId": "digit-sum-count",
   "promptKind": "exact-numeric",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 10,
   "startedAt": "2026-10-06T14:42:37.921Z",
   "finishedAt": "2026-10-06T14:42:44.947Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6408,
   "totalMs": 6804,
   "modelTurnMs": 6019,
   "inputTokens": 1917,
   "outputTokens": 725,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 452,
   "reasoningTokens": 722,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 3,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1",
   "answerNumber": 282
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 10,
   "startedAt": "2026-10-06T14:42:44.948Z",
   "finishedAt": "2026-10-06T14:42:52.697Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6020,
   "totalMs": 7050,
   "modelTurnMs": 6303,
   "inputTokens": 3915,
   "outputTokens": 815,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 694,
   "passed": false,
   "formatMiss": true,
   "answerCorrect": true,
   "checks": 0,
   "failures": [
    "parse: Unexpected token '`', \"```json\n{\n\"... is not valid JSON"
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 236,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R3",
   "failedOutputLineCount": 12
  },
  {
   "batch": "batch-2",
   "promptId": "order-json",
   "promptKind": "json-shape",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 10,
   "startedAt": "2026-10-06T14:42:52.698Z",
   "finishedAt": "2026-10-06T14:42:55.911Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1845,
   "totalMs": 2765,
   "modelTurnMs": 1893,
   "inputTokens": 2242,
   "outputTokens": 199,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 777,
   "reasoningTokens": 99,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 9,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 180,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 10,
   "startedAt": "2026-10-06T14:42:55.911Z",
   "finishedAt": "2026-10-06T14:43:01.252Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3729,
   "totalMs": 4895,
   "modelTurnMs": 4031,
   "inputTokens": 3757,
   "outputTokens": 462,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 334,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 309,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ],
   "answerId": "A6",
   "rawAnswerId": "R6"
  },
  {
   "batch": "batch-2",
   "promptId": "median-fix",
   "promptKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 10,
   "startedAt": "2026-10-06T14:43:01.252Z",
   "finishedAt": "2026-10-06T14:43:06.033Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3036,
   "totalMs": 4336,
   "modelTurnMs": 3494,
   "inputTokens": 2049,
   "outputTokens": 154,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 584,
   "reasoningTokens": 0,
   "passed": true,
   "formatMiss": false,
   "answerCorrect": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 325,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "answerId": "A1",
   "rawAnswerId": "R1"
  }
 ],
 "protocolNote": "The run protocol was declared before the first run. Its public summary is the Method section of the study page."
}
