{
 "kind": "jev-live-summary",
 "note": "Jev 1.13 (TypeSafe) called live over HTTPS on the 82 typed System One decisions the routing study uses: 3 repeats, 246 counted calls, one at a time, from one Mac over a home network. Client wall time per call, so the network is inside every latency; the API reports no server time. Cost per 1,000 decisions is a calculation from the reported input tokens and the published price.",
 "recordedAt": "2026-10-06",
 "run": {
  "startedAt": "2026-10-06T20:12:02.243Z",
  "endedAt": "2026-10-06T20:12:37.261Z",
  "plannedCalls": 246,
  "callsMade": 246,
  "reps": 3,
  "concurrency": 1,
  "timeoutMs": 10000,
  "requestedModel": "jev-1.13.0",
  "modelReportedByApi": [
   "jev-1.13.0"
  ],
  "endpointHost": "api.typesafe.ai",
  "machine": {
   "cpu": "Apple M3 Ultra",
   "cores": 28,
   "arch": "arm64",
   "node": "v25.2.1"
  },
  "stoppedEarly": null
 },
 "probe": {
  "counted": false,
  "status": 200,
  "latencyMs": 186.5,
  "inputTokens": 666,
  "outputTokens": 65
 },
 "decisionSet": {
  "decisions": 82,
  "scoredKeys": 194,
  "byPurpose": {
   "control.failure_class": {
    "cases": 32,
    "asked": 18
   },
   "frontdoor.intent": {
    "cases": 32,
    "asked": 20
   },
   "memory.is_rule": {
    "cases": 34,
    "asked": 12
   },
   "context.shape": {
    "cases": 68,
    "asked": 32
   }
  },
  "suiteVersions": {
   "control.failure_class": "failure-class-2026-10-05.1",
   "frontdoor.intent": "frontdoor-intent-2026-10-04.2",
   "memory.is_rule": "memory-is-rule-2026-10-04.2",
   "context.shape": "context-shape-2026-10-05.1"
  }
 },
 "accuracy": {
  "scoring": "Value-level, as the routing study: a key is correct when the choice is a labelled acceptable answer; a decision is exact when every scored key is correct; an errored call is incorrect. Wilson 95%.",
  "perRep": [
   {
    "rep": 1,
    "calls": 82,
    "okCalls": 82,
    "errors": 0,
    "exact": {
     "k": 74,
     "n": 82,
     "rate": 0.9024,
     "ci95": [
      0.8191,
      0.9497
     ]
    },
    "keys": {
     "k": 184,
     "n": 194,
     "rate": 0.9485,
     "ci95": [
      0.9077,
      0.9718
     ]
    }
   },
   {
    "rep": 2,
    "calls": 82,
    "okCalls": 82,
    "errors": 0,
    "exact": {
     "k": 73,
     "n": 82,
     "rate": 0.8902,
     "ci95": [
      0.8044,
      0.9412
     ]
    },
    "keys": {
     "k": 183,
     "n": 194,
     "rate": 0.9433,
     "ci95": [
      0.9013,
      0.968
     ]
    }
   },
   {
    "rep": 3,
    "calls": 82,
    "okCalls": 82,
    "errors": 0,
    "exact": {
     "k": 74,
     "n": 82,
     "rate": 0.9024,
     "ci95": [
      0.8191,
      0.9497
     ]
    },
    "keys": {
     "k": 185,
     "n": 194,
     "rate": 0.9536,
     "ci95": [
      0.9142,
      0.9754
     ]
    }
   }
  ],
  "pooled": {
   "calls": 246,
   "okCalls": 246,
   "errors": 0,
   "exact": {
    "k": 221,
    "n": 246,
    "rate": 0.8984,
    "ci95": [
     0.8543,
     0.9302
    ]
   },
   "keys": {
    "k": 552,
    "n": 582,
    "rate": 0.9485,
    "ci95": [
     0.9274,
     0.9637
    ]
   },
   "caseLevelCi95": {
    "exact": [
     0.8191,
     0.9497
    ],
    "keys": [
     0.9077,
     0.9718
    ],
    "note": "The same 82 decisions run in each rep, so pooled calls are not independent. This interval takes the pooled rate at n = 82 cases (194 keys)."
   }
  },
  "byPurpose": {
   "control.failure_class": {
    "decisions": 18,
    "calls": 54,
    "okCalls": 54,
    "errors": 0,
    "exact": {
     "k": 54,
     "n": 54,
     "rate": 1,
     "ci95": [
      0.9336,
      1
     ]
    },
    "keys": {
     "k": 54,
     "n": 54,
     "rate": 1,
     "ci95": [
      0.9336,
      1
     ]
    },
    "perRepExact": [
     18,
     18,
     18
    ]
   },
   "frontdoor.intent": {
    "decisions": 20,
    "calls": 60,
    "okCalls": 60,
    "errors": 0,
    "exact": {
     "k": 60,
     "n": 60,
     "rate": 1,
     "ci95": [
      0.9398,
      1
     ]
    },
    "keys": {
     "k": 60,
     "n": 60,
     "rate": 1,
     "ci95": [
      0.9398,
      1
     ]
    },
    "perRepExact": [
     20,
     20,
     20
    ]
   },
   "memory.is_rule": {
    "decisions": 12,
    "calls": 36,
    "okCalls": 36,
    "errors": 0,
    "exact": {
     "k": 36,
     "n": 36,
     "rate": 1,
     "ci95": [
      0.9036,
      1
     ]
    },
    "keys": {
     "k": 36,
     "n": 36,
     "rate": 1,
     "ci95": [
      0.9036,
      1
     ]
    },
    "perRepExact": [
     12,
     12,
     12
    ]
   },
   "context.shape": {
    "decisions": 32,
    "calls": 96,
    "okCalls": 96,
    "errors": 0,
    "exact": {
     "k": 71,
     "n": 96,
     "rate": 0.7396,
     "ci95": [
      0.6438,
      0.8169
     ]
    },
    "keys": {
     "k": 402,
     "n": 432,
     "rate": 0.9306,
     "ci95": [
      0.9026,
      0.9509
     ]
    },
    "perRepExact": [
     24,
     23,
     24
    ]
   }
  },
  "stability": {
   "decisions": 82,
   "alwaysCorrect": 73,
   "neverCorrect": 8,
   "mixed": 1,
   "identicalAnswersAllReps": 73,
   "mixedCases": [
    {
     "id": "context.shape/cont-mid-implementation",
     "correctByRep": [
      true,
      false,
      true
     ]
    }
   ],
   "neverCorrectCases": [
    "context.shape/new-design-decision",
    "context.shape/cont-near-done",
    "context.shape/clarify-open-questions",
    "context.shape/ml-cont-no-summary",
    "context.shape/ml-cont-failing-suite",
    "context.shape/asked-clarify-limit",
    "context.shape/asked-followup-notes",
    "context.shape/asked-clarify-branch"
   ]
  }
 },
 "latencyMs": {
  "definition": "Client wall time per call: performance.now() from before fetch to after the response body was read (includes network from this Mac, TLS only on the cold call, server queue and inference, download). Quantiles by linear interpolation.",
  "calls": {
   "n": 246,
   "mean": 142.1,
   "median": 136.5,
   "p90": 171,
   "p95": 195.7,
   "min": 100.9,
   "max": 297.3
  },
  "callsWithoutColdFirst": {
   "n": 245,
   "mean": 141.8,
   "median": 136.4,
   "p90": 170.5,
   "p95": 193.2,
   "min": 100.9,
   "max": 297.3
  },
  "timeToHeaders": {
   "n": 246,
   "mean": 141.8,
   "median": 136.2,
   "p90": 170.8,
   "p95": 195.5,
   "min": 100.7,
   "max": 297.1
  },
  "timeToHeadersWithoutColdFirst": {
   "n": 245,
   "mean": 141.5,
   "median": 136.1,
   "p90": 170.2,
   "p95": 193,
   "min": 100.7,
   "max": 297.1
  },
  "coldFirstCall": {
   "latencyMs": 224.725,
   "timeToHeadersMs": 222.436,
   "ok": true,
   "note": "First counted call of the run, new process, fresh DNS/TCP/TLS from this client."
  },
  "probeNotCounted": {
   "latencyMs": 186.472,
   "timeToHeadersMs": 184.467,
   "note": "Uncounted control call in its own process, also a fresh connection."
  },
  "perRep": [
   {
    "rep": 1,
    "n": 81,
    "mean": 136.6,
    "median": 130.3,
    "p90": 161.1,
    "p95": 179.8,
    "min": 109.6,
    "max": 232.6,
    "note": "cold first call excluded"
   },
   {
    "rep": 2,
    "n": 82,
    "mean": 146.7,
    "median": 141.7,
    "p90": 172.5,
    "p95": 195.6,
    "min": 108.2,
    "max": 297.3
   },
   {
    "rep": 3,
    "n": 82,
    "mean": 141.9,
    "median": 137.1,
    "p90": 170,
    "p95": 192.2,
    "min": 100.9,
    "max": 239
   }
  ],
  "byPurpose": {
   "control.failure_class": {
    "n": 53,
    "mean": 138.2,
    "median": 130.8,
    "p90": 168.8,
    "p95": 174.4,
    "min": 108.2,
    "max": 233.8
   },
   "frontdoor.intent": {
    "n": 60,
    "mean": 141.5,
    "median": 135.5,
    "p90": 174.6,
    "p95": 180.5,
    "min": 100.9,
    "max": 286.8
   },
   "memory.is_rule": {
    "n": 36,
    "mean": 147.6,
    "median": 138.7,
    "p90": 178.7,
    "p95": 206.7,
    "min": 109.6,
    "max": 297.3
   },
   "context.shape": {
    "n": 96,
    "mean": 141.7,
    "median": 137.4,
    "p90": 161.6,
    "p95": 193.7,
    "min": 112.8,
    "max": 239
   }
  },
  "slowestCalls": [
   {
    "n": 126,
    "rep": 2,
    "id": "memory.is_rule/docstrings-norm",
    "latencyMs": 297.348
   },
   {
    "n": 107,
    "rep": 2,
    "id": "frontdoor.intent/hold-everything",
    "latencyMs": 286.83
   },
   {
    "n": 111,
    "rep": 2,
    "id": "frontdoor.intent/also-canada",
    "latencyMs": 274.446
   },
   {
    "n": 235,
    "rep": 3,
    "id": "context.shape/ml-cont-failing-suite",
    "latencyMs": 238.971
   },
   {
    "n": 95,
    "rep": 2,
    "id": "control.failure_class/go-panic",
    "latencyMs": 233.762
   }
  ],
  "failedCallsLatency": [],
  "serverTiming": "not reported: the API sends no timing header and no timing field (response keys: model, answers, usage)."
 },
 "errors": {
  "count": 0,
  "byKind": {},
  "statuses": {
   "200": 246
  },
  "retried": 0,
  "stoppedEarly": null
 },
 "usageAndCost": {
  "fieldsReported": [
   "input_tokens",
   "output_tokens"
  ],
  "callsWithUsage": 246,
  "inputTokensPerDecision": {
   "mean": 802.8,
   "median": 666.5,
   "min": 475,
   "max": 1415
  },
  "inputTokensPerRep": [
   65826,
   65826,
   65826
  ],
  "outputTokensPerDecision": {
   "mean": 148.4,
   "note": "Reported by the API; output tokens are free at the published price."
  },
  "providerReportedCost": "not reported (no cost field in the response usage object)",
  "publishedPrice": {
   "inputUsdPerMTok": 0.042,
   "outputUsdPerMTok": 0,
   "source": "docs.typesafe.ai/models read 2026-10-06 ($42 per Btok input, output free); same as the product price table"
  },
  "costPer1000DecisionsUsd": {
   "value": 0.03372,
   "kind": "calculation",
   "formula": "mean reported input_tokens per decision x 1,000 x $0.042 / 1,000,000; output tokens free"
  },
  "costPerDecisionUsd": 0.00003372
 },
 "recordedProductionRun": {
  "recordedRun": {
   "calls": 82,
   "exact": {
    "k": 74,
    "n": 82,
    "rate": 0.9024,
    "ci95": [
     0.8191,
     0.9497
    ]
   },
   "keys": {
    "k": 185,
    "n": 194,
    "rate": 0.9536,
    "ci95": [
     0.9142,
     0.9754
    ]
   },
   "inputTokens": 65826,
   "inputTokensPerDecision": 802.8,
   "costMicroUsdProviderReported": 2765,
   "latency": "not recorded",
   "note": "The recorded report keeps the production confidence path; correctness here is value-level, as in the routing study."
  },
  "liveVsRecorded": [
   {
    "rep": 1,
    "decisionsWithIdenticalAnswers": 76,
    "ofDecisions": 82,
    "keysWithIdenticalAnswer": 243,
    "ofKeys": 249,
    "decisionsWithSameCorrectness": 82,
    "decisionsWhereCorrectnessChanged": []
   },
   {
    "rep": 2,
    "decisionsWithIdenticalAnswers": 75,
    "ofDecisions": 82,
    "keysWithIdenticalAnswer": 242,
    "ofKeys": 249,
    "decisionsWithSameCorrectness": 81,
    "decisionsWhereCorrectnessChanged": [
     {
      "id": "context.shape/cont-mid-implementation",
      "recordedCorrect": true,
      "liveCorrect": false
     }
    ]
   },
   {
    "rep": 3,
    "decisionsWithIdenticalAnswers": 76,
    "ofDecisions": 82,
    "keysWithIdenticalAnswer": 243,
    "ofKeys": 249,
    "decisionsWithSameCorrectness": 82,
    "decisionsWhereCorrectnessChanged": []
   }
  ],
  "note": "Live reps are separate samples of the same model on the same requests about one day after the recorded run. keysWithIdenticalAnswer counts every key asked (249, including keys without a label), decisionsWithIdenticalAnswers needs all asked keys equal. The live reps scored 74, 73 and 74 exact against 74 in the recorded run: a difference of one decision is within the repeat-to-repeat spread of the live reps."
 },
 "repoRunnerCrossCheck": {
  "source": "repo-runner-check.json (calls.jsonl replayed through the repo runDecisionEval)",
  "reps": [
   {
    "rep": 1,
    "harnessExact": 74,
    "repoExact": 74,
    "harnessKeys": 184,
    "repoKeys": 184,
    "repoCases": 82,
    "repoKeysScored": 194,
    "agrees": true
   },
   {
    "rep": 2,
    "harnessExact": 73,
    "repoExact": 73,
    "harnessKeys": 183,
    "repoKeys": 183,
    "repoCases": 82,
    "repoKeysScored": 194,
    "agrees": true
   },
   {
    "rep": 3,
    "harnessExact": 74,
    "repoExact": 74,
    "harnessKeys": 185,
    "repoKeys": 185,
    "repoCases": 82,
    "repoKeysScored": 194,
    "agrees": true
   }
  ]
 },
 "caveats": [
  "Home advantage: the case sets and question wording were revised in fix waves against Jev answers on 2026-10-04 and 2026-10-05.",
  "82 decisions in four small hand-labelled sets; intervals are wide, and the pooled 246-call interval understates the uncertainty because the same 82 decisions repeat.",
  "One client machine, one network path, a 35-second window. Latency is client wall time to a hosted API from a home network, not model compute time; a server in the same region would see less.",
  "Server-side timing is not reported by the API.",
  "The comparison rows come from other studies, with a CLI in the path for the LLM routers; they are not a same-harness head-to-head."
 ],
 "protocolNote": "The run protocol was declared before the first run. Its public summary is the Method section of the study page."
}
