{
 "note": "Routing overhead: what delay and cost each kind of router adds before a call’s real work starts. Deterministic policy timed in process; LLM routers from the recorded routing runs; CLI start-up measured with a minimal prompt; per-task figures are calculations.",
 "generatedAt": "2026-10-06T12:38:48.929Z",
 "question": "What delay and what cost does each kind of router add before the real work of a call starts?",
 "perTask": {
  "runs": 48,
  "emptyRunsExcluded": 2,
  "modelCalls": {
   "median": 49.5,
   "mean": 51.4,
   "min": 13,
   "max": 73
  },
  "systemOneDecisions": {
   "median": 7,
   "mean": 8.7,
   "min": 2,
   "max": 24
  },
  "workCostUsd": {
   "median": 3.026,
   "mean": 3.19
  },
  "wallMinutes": {
   "median": 10.3
  },
  "routingDecisionRows": 0,
  "systemOneLatencyByArm": {
   "rule": {
    "n": 419,
    "p50Ms": 1,
    "p95Ms": 2,
    "maxMs": 3
   }
  },
  "note": "Bench runs had model routing off (0 routing_decisions rows), so every model call is counted as one decision a router would make. System One decisions were all answered by the rule arm."
 },
 "caveats": [
  "LLM router latency is CLI wall time (one call at a time); a direct API call would skip the CLI overhead, so the API-only column is shown too.",
  "Haiku 4.5 ran with the CLI default extended thinking, which makes it slower than Sonnet 5.5 at effort low here.",
  "The per-1,000-task figures are calculations: decisions per task from recorded runs times the measured or recorded per-decision numbers; decisions are assumed sequential.",
  "Work cost per task is the cost the bench runs recorded for their model calls (claude-code provider, estimated at list price), Sonnet 5.5 for every call with routing off.",
  "Jev, OpenRouter and other hosted routers were not timed: no key in this environment.",
  "The microbenchmark times the pure decision; database reads and the decision record write in production are not included (the recorded System One rule arm, which includes them at millisecond resolution, is shown separately)."
 ],
 "records": [
  {
   "type": "router-overhead",
   "id": "claude-haiku",
   "label": "Claude Haiku 4.5",
   "kind": "LLM router via agent CLI",
   "source": "recorded 2026-10-05 routing study (82 calls, one at a time)",
   "calls": 82,
   "okCalls": 82,
   "failedCalls": 0,
   "latencyMs": {
    "wall": {
     "n": 82,
     "p50": 12543,
     "p95": 34481,
     "min": 5857,
     "max": 51278
    },
    "modelApi": {
     "n": 82,
     "p50": 10508,
     "p95": 32132,
     "min": 4328,
     "max": 49487
    },
    "cliOverhead": {
     "n": 82,
     "p50": 1698,
     "p95": 2677,
     "min": 1356,
     "max": 3872
    }
   },
   "costPerDecisionUsd": 0.008924,
   "costPer1000DecisionsUsd": 8.924,
   "costSource": "list price x CLI-reported tokens"
  },
  {
   "type": "router-overhead",
   "id": "claude-sonnet",
   "label": "Claude Sonnet 5.5",
   "kind": "LLM router via agent CLI",
   "source": "recorded 2026-10-05 routing study (82 calls, one at a time)",
   "calls": 82,
   "okCalls": 82,
   "failedCalls": 0,
   "latencyMs": {
    "wall": {
     "n": 82,
     "p50": 2597,
     "p95": 4298,
     "min": 1993,
     "max": 5583
    },
    "modelApi": {
     "n": 82,
     "p50": 1596,
     "p95": 2583,
     "min": 1060,
     "max": 4757
    },
    "cliOverhead": {
     "n": 82,
     "p50": 973,
     "p95": 1277,
     "min": 826,
     "max": 3673
    }
   },
   "costPerDecisionUsd": 0.004996,
   "costPer1000DecisionsUsd": 4.996,
   "costSource": "list price x CLI-reported tokens"
  },
  {
   "type": "router-overhead",
   "id": "jev",
   "label": "Jev 1.13 (TypeSafe)",
   "kind": "hosted decision model",
   "source": "recorded production run 2026-10-05 (82 calls)",
   "calls": 82,
   "okCalls": 82,
   "failedCalls": 0,
   "latencyMs": null,
   "latencyNote": "Not measured: no key in this environment.",
   "costPerDecisionUsd": 0.0000337,
   "costPer1000DecisionsUsd": 0.0337,
   "costSource": "provider-reported"
  },
  {
   "type": "router-overhead",
   "id": "policy",
   "label": "Platform routing policy (deterministic)",
   "kind": "in-process rules",
   "source": "microbenchmark on this Mac (Apple M3 Ultra, Node v25.2.1), 2026-10-06",
   "calls": 20000,
   "okCalls": 20000,
   "failedCalls": 0,
   "latencyUs": {
    "p50": 1.42,
    "p95": 2.33,
    "p99": 3.04,
    "max": 2538.21,
    "mean": 1.84,
    "batchMean": 2.13,
    "warmup": 5000,
    "contexts": 64,
    "timerResolution": 0.041
   },
   "latencyUsWithPolicyMerge": {
    "p50": 1.62,
    "p95": 2.54,
    "p99": 4.71,
    "max": 37713.12,
    "mean": 4.09,
    "batchMean": 2.16
   },
   "decisionsPerSecond": 469409,
   "costPerDecisionUsd": 0,
   "costPer1000DecisionsUsd": 0,
   "costSource": "no model call",
   "outOfScope": [
    "DB reads routeCall makes for history, overrides and learned adjustments",
    "routing_decisions insert",
    "brain.decide (off by default)"
   ]
  },
  {
   "type": "router-overhead",
   "id": "system-one-rules",
   "label": "System One rule arm (recorded in bench runs)",
   "kind": "in-process rules + record write",
   "source": "decision_records in 48 bench runs (millisecond resolution)",
   "calls": 419,
   "okCalls": 419,
   "failedCalls": 0,
   "latencyMs": {
    "recorded": {
     "n": 419,
     "p50": 1,
     "p95": 2,
     "max": 3
    }
   },
   "costPerDecisionUsd": 0,
   "costPer1000DecisionsUsd": 0,
   "costSource": "no model call"
  },
  {
   "type": "router-overhead",
   "id": "openrouter-auto",
   "label": "OpenRouter Auto Router / cheaper hosted inference",
   "kind": "hosted LLM router / gateway",
   "measured": null,
   "notMeasuredReason": "Not measured: no key in the environment. The harness is ready and runs when a key is set."
  },
  {
   "type": "router-overhead",
   "id": "clef-local",
   "label": "Clef / Clef-Flash (local)",
   "kind": "local router model",
   "measured": null,
   "notMeasuredReason": "Not measured: no local server running."
  },
  {
   "type": "cli-startup",
   "id": "claude-cli",
   "label": "Claude Code CLI (Haiku 4.5, isolated flags)",
   "source": "this Mac, 2026-10-06, minimal prompt, one at a time",
   "runs": 5,
   "okRuns": 5,
   "failedRuns": 0,
   "firstEventMs": {
    "n": 5,
    "p50": 563,
    "p95": 726,
    "min": 519,
    "max": 726
   },
   "firstModelEventMs": {
    "n": 5,
    "p50": 1461,
    "p95": 2308,
    "min": 1206,
    "max": 2308
   },
   "wallMs": {
    "n": 5,
    "p50": 2529,
    "p95": 3382,
    "min": 2273,
    "max": 3382
   },
   "modelApiMs": {
    "n": 5,
    "p50": 908,
    "p95": 1598,
    "min": 718,
    "max": 1598
   },
   "harnessMs": {
    "n": 5,
    "p50": 1690,
    "p95": 1811,
    "min": 1533,
    "max": 1811
   },
   "inputTokensPerCall": 6761,
   "note": "Harness = wall time minus the CLI-reported API time. Includes process start, init and a post-turn summary step before exit."
  },
  {
   "type": "cli-startup",
   "id": "codex-cli",
   "label": "Codex CLI (exec --json, default model)",
   "source": "this Mac, 2026-10-06, minimal prompt, one at a time",
   "runs": 5,
   "okRuns": 5,
   "failedRuns": 0,
   "firstEventMs": {
    "n": 5,
    "p50": 489,
    "p95": 1304,
    "min": 354,
    "max": 1304
   },
   "firstModelEventMs": {
    "n": 5,
    "p50": 5059,
    "p95": 5478,
    "min": 4391,
    "max": 5478
   },
   "wallMs": {
    "n": 5,
    "p50": 5999,
    "p95": 6506,
    "min": 5367,
    "max": 6506
   },
   "modelApiMs": null,
   "harnessMs": null,
   "inputTokensPerCall": 17051,
   "note": "The Codex CLI reports no API time; model time and harness time are not separable. Input tokens include the CLI system prompt for a five-token answer.",
   "cacheReadTokensPerCall": 13184
  },
  {
   "type": "overhead-per-1000-tasks",
   "kind": "calculation",
   "router": "policy",
   "routerLabel": "Deterministic policy",
   "scope": "every-model-call",
   "scopeLabel": "one routing decision per model call",
   "decisionsPerTask": 49.5,
   "addedMsPerTaskP50": 0.0703,
   "addedSecondsPerTaskP50": 0.0001,
   "addedSecondsPerTaskP95Each": 0.0001,
   "addedSecondsPerTaskApiOnlyP50": null,
   "addedHoursPer1000TasksP50": 0,
   "addedCostPer1000TasksUsd": 0,
   "shareOfMedianWorkCost": 0,
   "latencyNote": "Sequential decisions on the critical path (upper bound). P95 column multiplies the per-decision p95; it is not a task p95."
  },
  {
   "type": "overhead-per-1000-tasks",
   "kind": "calculation",
   "router": "jev",
   "routerLabel": "Jev 1.13 (TypeSafe)",
   "scope": "every-model-call",
   "scopeLabel": "one routing decision per model call",
   "decisionsPerTask": 49.5,
   "addedMsPerTaskP50": null,
   "addedSecondsPerTaskP50": null,
   "addedSecondsPerTaskP95Each": null,
   "addedSecondsPerTaskApiOnlyP50": null,
   "addedHoursPer1000TasksP50": null,
   "addedCostPer1000TasksUsd": 1.67,
   "shareOfMedianWorkCost": 0.0006,
   "latencyNote": "Not measured: no key in this environment."
  },
  {
   "type": "overhead-per-1000-tasks",
   "kind": "calculation",
   "router": "claude-sonnet",
   "routerLabel": "Claude Sonnet 5.5 via CLI",
   "scope": "every-model-call",
   "scopeLabel": "one routing decision per model call",
   "decisionsPerTask": 49.5,
   "addedMsPerTaskP50": 128551.5,
   "addedSecondsPerTaskP50": 128.5515,
   "addedSecondsPerTaskP95Each": 212.751,
   "addedSecondsPerTaskApiOnlyP50": 79,
   "addedHoursPer1000TasksP50": 35.7088,
   "addedCostPer1000TasksUsd": 247.3,
   "shareOfMedianWorkCost": 0.0817,
   "latencyNote": "Sequential decisions on the critical path (upper bound). P95 column multiplies the per-decision p95; it is not a task p95."
  },
  {
   "type": "overhead-per-1000-tasks",
   "kind": "calculation",
   "router": "claude-haiku",
   "routerLabel": "Claude Haiku 4.5 via CLI (thinking on)",
   "scope": "every-model-call",
   "scopeLabel": "one routing decision per model call",
   "decisionsPerTask": 49.5,
   "addedMsPerTaskP50": 620878.5,
   "addedSecondsPerTaskP50": 620.8785,
   "addedSecondsPerTaskP95Each": 1706.8095,
   "addedSecondsPerTaskApiOnlyP50": 520.15,
   "addedHoursPer1000TasksP50": 172.4663,
   "addedCostPer1000TasksUsd": 441.74,
   "shareOfMedianWorkCost": 0.146,
   "latencyNote": "Sequential decisions on the critical path (upper bound). P95 column multiplies the per-decision p95; it is not a task p95."
  },
  {
   "type": "overhead-per-1000-tasks",
   "kind": "calculation",
   "router": "policy",
   "routerLabel": "Deterministic policy",
   "scope": "system-one-only",
   "scopeLabel": "only the System One decisions a task made",
   "decisionsPerTask": 7,
   "addedMsPerTaskP50": 0.0099,
   "addedSecondsPerTaskP50": 0,
   "addedSecondsPerTaskP95Each": 0,
   "addedSecondsPerTaskApiOnlyP50": null,
   "addedHoursPer1000TasksP50": 0,
   "addedCostPer1000TasksUsd": 0,
   "shareOfMedianWorkCost": 0,
   "latencyNote": "Sequential decisions on the critical path (upper bound). P95 column multiplies the per-decision p95; it is not a task p95."
  },
  {
   "type": "overhead-per-1000-tasks",
   "kind": "calculation",
   "router": "jev",
   "routerLabel": "Jev 1.13 (TypeSafe)",
   "scope": "system-one-only",
   "scopeLabel": "only the System One decisions a task made",
   "decisionsPerTask": 7,
   "addedMsPerTaskP50": null,
   "addedSecondsPerTaskP50": null,
   "addedSecondsPerTaskP95Each": null,
   "addedSecondsPerTaskApiOnlyP50": null,
   "addedHoursPer1000TasksP50": null,
   "addedCostPer1000TasksUsd": 0.24,
   "shareOfMedianWorkCost": 0.0001,
   "latencyNote": "Not measured: no key in this environment."
  },
  {
   "type": "overhead-per-1000-tasks",
   "kind": "calculation",
   "router": "claude-sonnet",
   "routerLabel": "Claude Sonnet 5.5 via CLI",
   "scope": "system-one-only",
   "scopeLabel": "only the System One decisions a task made",
   "decisionsPerTask": 7,
   "addedMsPerTaskP50": 18179,
   "addedSecondsPerTaskP50": 18.179,
   "addedSecondsPerTaskP95Each": 30.086,
   "addedSecondsPerTaskApiOnlyP50": 11.17,
   "addedHoursPer1000TasksP50": 5.0497,
   "addedCostPer1000TasksUsd": 34.97,
   "shareOfMedianWorkCost": 0.0116,
   "latencyNote": "Sequential decisions on the critical path (upper bound). P95 column multiplies the per-decision p95; it is not a task p95."
  },
  {
   "type": "overhead-per-1000-tasks",
   "kind": "calculation",
   "router": "claude-haiku",
   "routerLabel": "Claude Haiku 4.5 via CLI (thinking on)",
   "scope": "system-one-only",
   "scopeLabel": "only the System One decisions a task made",
   "decisionsPerTask": 7,
   "addedMsPerTaskP50": 87801,
   "addedSecondsPerTaskP50": 87.801,
   "addedSecondsPerTaskP95Each": 241.367,
   "addedSecondsPerTaskApiOnlyP50": 73.56,
   "addedHoursPer1000TasksP50": 24.3892,
   "addedCostPer1000TasksUsd": 62.47,
   "shareOfMedianWorkCost": 0.0206,
   "latencyNote": "Sequential decisions on the critical path (upper bound). P95 column multiplies the per-decision p95; it is not a task p95."
  }
 ],
 "runsPerTask": [
  {
   "modelCalls": 65,
   "systemOneDecisions": 14,
   "workCostUsd": 4.0029,
   "wallMinutes": 13.77
  },
  {
   "modelCalls": 64,
   "systemOneDecisions": 8,
   "workCostUsd": 3.3499,
   "wallMinutes": 10.11
  },
  {
   "modelCalls": 48,
   "systemOneDecisions": 15,
   "workCostUsd": 2.0315,
   "wallMinutes": 4.48
  },
  {
   "modelCalls": 34,
   "systemOneDecisions": 6,
   "workCostUsd": 1.5186,
   "wallMinutes": 4.39
  },
  {
   "modelCalls": 46,
   "systemOneDecisions": 7,
   "workCostUsd": 2.5122,
   "wallMinutes": 7.66
  },
  {
   "modelCalls": 43,
   "systemOneDecisions": 6,
   "workCostUsd": 3.0847,
   "wallMinutes": 9.45
  },
  {
   "modelCalls": 38,
   "systemOneDecisions": 7,
   "workCostUsd": 2.2653,
   "wallMinutes": 6.95
  },
  {
   "modelCalls": 43,
   "systemOneDecisions": 7,
   "workCostUsd": 2.0633,
   "wallMinutes": 5.43
  },
  {
   "modelCalls": 54,
   "systemOneDecisions": 6,
   "workCostUsd": 2.984,
   "wallMinutes": 10.14
  },
  {
   "modelCalls": 43,
   "systemOneDecisions": 6,
   "workCostUsd": 2.1485,
   "wallMinutes": 6.67
  },
  {
   "modelCalls": 43,
   "systemOneDecisions": 6,
   "workCostUsd": 2.2918,
   "wallMinutes": 6.49
  },
  {
   "modelCalls": 54,
   "systemOneDecisions": 7,
   "workCostUsd": 3.0285,
   "wallMinutes": 11.2
  },
  {
   "modelCalls": 38,
   "systemOneDecisions": 6,
   "workCostUsd": 1.7363,
   "wallMinutes": 5.3
  },
  {
   "modelCalls": 47,
   "systemOneDecisions": 7,
   "workCostUsd": 2.7056,
   "wallMinutes": 6.88
  },
  {
   "modelCalls": 47,
   "systemOneDecisions": 6,
   "workCostUsd": 2.7991,
   "wallMinutes": 37.82
  },
  {
   "modelCalls": 58,
   "systemOneDecisions": 6,
   "workCostUsd": 2.9948,
   "wallMinutes": 8.61
  },
  {
   "modelCalls": 62,
   "systemOneDecisions": 7,
   "workCostUsd": 4.0978,
   "wallMinutes": 14.62
  },
  {
   "modelCalls": 59,
   "systemOneDecisions": 5,
   "workCostUsd": 3.6725,
   "wallMinutes": 10.68
  },
  {
   "modelCalls": 42,
   "systemOneDecisions": 6,
   "workCostUsd": 1.8051,
   "wallMinutes": 4.94
  },
  {
   "modelCalls": 59,
   "systemOneDecisions": 6,
   "workCostUsd": 2.9043,
   "wallMinutes": 10.52
  },
  {
   "modelCalls": 47,
   "systemOneDecisions": 7,
   "workCostUsd": 2.554,
   "wallMinutes": 7.2
  },
  {
   "modelCalls": 37,
   "systemOneDecisions": 6,
   "workCostUsd": 1.613,
   "wallMinutes": 12.64
  },
  {
   "modelCalls": 42,
   "systemOneDecisions": 7,
   "workCostUsd": 2.276,
   "wallMinutes": 7.4
  },
  {
   "modelCalls": 46,
   "systemOneDecisions": 11,
   "workCostUsd": 3.2854,
   "wallMinutes": 12.69
  },
  {
   "modelCalls": 53,
   "systemOneDecisions": 11,
   "workCostUsd": 3.9968,
   "wallMinutes": 11.28
  },
  {
   "modelCalls": 48,
   "systemOneDecisions": 10,
   "workCostUsd": 3.8091,
   "wallMinutes": 12.07
  },
  {
   "modelCalls": 35,
   "systemOneDecisions": 10,
   "workCostUsd": 2.9583,
   "wallMinutes": 8.76
  },
  {
   "modelCalls": 53,
   "systemOneDecisions": 8,
   "workCostUsd": 3.0234,
   "wallMinutes": 9.05
  },
  {
   "modelCalls": 64,
   "systemOneDecisions": 12,
   "workCostUsd": 3.6441,
   "wallMinutes": 40.79
  },
  {
   "modelCalls": 66,
   "systemOneDecisions": 23,
   "workCostUsd": 3.8958,
   "wallMinutes": 32.69
  },
  {
   "modelCalls": 57,
   "systemOneDecisions": 21,
   "workCostUsd": 4.0911,
   "wallMinutes": 15.97
  },
  {
   "modelCalls": 50,
   "systemOneDecisions": 8,
   "workCostUsd": 3.2297,
   "wallMinutes": 34.98
  },
  {
   "modelCalls": 44,
   "systemOneDecisions": 6,
   "workCostUsd": 2.4348,
   "wallMinutes": 8.27
  },
  {
   "modelCalls": 73,
   "systemOneDecisions": 8,
   "workCostUsd": 4.8537,
   "wallMinutes": 14.97
  },
  {
   "modelCalls": 71,
   "systemOneDecisions": 8,
   "workCostUsd": 4.8943,
   "wallMinutes": 13.24
  },
  {
   "modelCalls": 43,
   "systemOneDecisions": 8,
   "workCostUsd": 3.4259,
   "wallMinutes": 8.52
  },
  {
   "modelCalls": 29,
   "systemOneDecisions": 2,
   "workCostUsd": 2.4395,
   "wallMinutes": 6.83
  },
  {
   "modelCalls": 52,
   "systemOneDecisions": 7,
   "workCostUsd": 3.5339,
   "wallMinutes": 10.89
  },
  {
   "modelCalls": 67,
   "systemOneDecisions": 24,
   "workCostUsd": 4.8274,
   "wallMinutes": 16.52
  },
  {
   "modelCalls": 49,
   "systemOneDecisions": 8,
   "workCostUsd": 2.6926,
   "wallMinutes": 9.33
  },
  {
   "modelCalls": 13,
   "systemOneDecisions": 2,
   "workCostUsd": 0.8898,
   "wallMinutes": 1.32
  },
  {
   "modelCalls": 72,
   "systemOneDecisions": 7,
   "workCostUsd": 5.3769,
   "wallMinutes": 15.9
  },
  {
   "modelCalls": 71,
   "systemOneDecisions": 7,
   "workCostUsd": 4.8061,
   "wallMinutes": 20.58
  },
  {
   "modelCalls": 28,
   "systemOneDecisions": 5,
   "workCostUsd": 1.7479,
   "wallMinutes": 6.04
  },
  {
   "modelCalls": 0,
   "workCostUsd": null,
   "wallMinutes": null
  },
  {
   "modelCalls": 0,
   "workCostUsd": null,
   "wallMinutes": null
  },
  {
   "modelCalls": 67,
   "systemOneDecisions": 11,
   "workCostUsd": 4.8794,
   "wallMinutes": 17.41
  },
  {
   "modelCalls": 68,
   "systemOneDecisions": 9,
   "workCostUsd": 4.6258,
   "wallMinutes": 18.99
  },
  {
   "modelCalls": 73,
   "systemOneDecisions": 15,
   "workCostUsd": 4.7365,
   "wallMinutes": 22.11
  },
  {
   "modelCalls": 62,
   "systemOneDecisions": 14,
   "workCostUsd": 4.574,
   "wallMinutes": 19.41
  }
 ],
 "protocolNote": "The run protocol was declared before the first run. Its public summary is the Method section of the study page."
}
