{"$k":["id","title","columns","rows"],"$r":[["latency-budget-measured-steps","Every step and the time it was measured to take",{"$k":["key","label","unit"],"$r":[["step","Step","text"],["covers","What the time covers","text"],["route","Route","text"],["n","Runs","count"],["p50","Median","ms"],["range","Observed range (not an interval)","text"],["slow","Slow end","ms"],["slowKind","Slow end is","text"],["origin","Where the time comes from","text"]]},{"$k":["step","covers","route","n","p50","range","slow","slowKind","origin"],"$r":[["Deterministic routing policy (Agent, in process)","pure in-process policy decision; no database work","in process, no model call",20000,0.00142,"Minimum not retained to 2.5 ms (n = 20,000)",0.00233,"95th percentile","Routing overhead run: in-process policy timing, p50 and p95 (µs)"],["Rule-based System One decision (record write excluded)","whole decision, request to answer","in process, record write excluded",419,1,"Minimum not retained to 3 ms (n = 419)",2,"95th percentile","Routing overhead run: recorded System One decisions (millisecond resolution)"],["Jev 1.13 (TypeSafe, direct HTTPS)","whole decision, request to answer","direct HTTPS, home network",246,136.5,"100.9 ms to 297.3 ms (n = 246)",195.7,"95th percentile","Jev live run: client wall time per call"],["OpenAI API · GPT-6 Luna · none (first output, one-line answer)","time to first useful output","OpenAI API",5,823,"506 ms to 1,371 ms (n = 5)",1371,"slowest run","Provider explorer receipts: first useful output, matched cohort, fixed exact reply"],["OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)","time to first useful output","OpenAI API",5,873,"835 ms to 1,742 ms (n = 5)",1742,"slowest run","Provider explorer receipts: first useful output, matched cohort, fixed exact reply"],["Claude Code · Claude Fable 5.1 (first output, 5 short tasks)","time to first useful output","Claude Code CLI",15,1196,"947 ms to 7,902 ms (n = 15)",7902,"slowest run","Provider head-to-head receipts: first useful output, configuration with the lowest observed median"],["OpenAI API · GPT-6.1 Sol · high (first output, one-line answer)","time to first useful output","OpenAI API",5,1341,"1,259 ms to 2,115 ms (n = 5)",2115,"slowest run","Provider explorer receipts: first useful output, matched cohort, fixed exact reply"],["Claude Code · Claude Haiku 4.5 (start-up, first model output)","time to first useful output","Claude Code CLI, one-word prompt",5,1461,"1,206 ms to 2,308 ms (n = 5)",2308,"slowest run","Routing overhead run: CLI start-up, time to the first model output"],["Claude Sonnet 5.5 (router, effort low, via Claude Code)","whole decision, request to answer","Claude Code CLI",82,2598,"1,993 ms to 5,583 ms (n = 82)",4298,"95th percentile","Routing run: median and interpolated p95; overhead extract: n and observed limits"],["Codex CLI · GPT-6 Luna · none (first output, one-line answer)","time to first useful output","Codex CLI",5,2787,"2,461 ms to 3,418 ms (n = 5)",3418,"slowest run","Provider explorer receipts: first useful output, matched cohort, fixed exact reply"],["Codex CLI · GPT-6.1 Sol · low (first output, one-line answer)","time to first useful output","Codex CLI",5,3753,"3,435 ms to 4,103 ms (n = 5)",4103,"slowest run","Provider explorer receipts: first useful output, matched cohort, fixed exact reply"],["Codex CLI · GPT-6.1 Sol · high (first output, one-line answer)","time to first useful output","Codex CLI",5,3786,"3,366 ms to 4,296 ms (n = 5)",4296,"slowest run","Provider explorer receipts: first useful output, matched cohort, fixed exact reply"],["Codex CLI (start-up, first model output)","time to first useful output","Codex CLI, one-word prompt",5,5059,"4,391 ms to 5,478 ms (n = 5)",5478,"slowest run","Routing overhead run: CLI start-up, time to the first model output"],["Claude Haiku 4.5 (router, thinking on, via Claude Code)","whole decision, request to answer","Claude Code CLI",82,12674,"5,857 ms to 51.28 s (n = 82)",34413,"95th percentile","Routing run: median and interpolated p95; overhead extract: n and observed limits"]]}],["latency-budget-fits","Which steps fit each assumed budget (calculation)",{"$k":["key","label","unit"],"$r":[["step","Step","text"],["b300","Fits 300 ms","text"],["b800","Fits 800 ms","text"],["b1500","Fits 1,500 ms","text"]]},{"$k":["step","b300","b800","b1500"],"$r":[["Deterministic routing policy (Agent, in process)","median and slow end","median and slow end","median and slow end"],["Rule-based System One decision (record write excluded)","median and slow end","median and slow end","median and slow end"],["Jev 1.13 (TypeSafe, direct HTTPS)","median and slow end","median and slow end","median and slow end"],["OpenAI API · GPT-6 Luna · none (first output, one-line answer)","neither","neither","median and slow end"],["OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)","neither","neither","median only"],["Claude Code · Claude Fable 5.1 (first output, 5 short tasks)","neither","neither","median only"],["OpenAI API · GPT-6.1 Sol · high (first output, one-line answer)","neither","neither","median only"],["Claude Code · Claude Haiku 4.5 (start-up, first model output)","neither","neither","median only"],["Claude Sonnet 5.5 (router, effort low, via Claude Code)","neither","neither","neither"],["Codex CLI · GPT-6 Luna · none (first output, one-line answer)","neither","neither","neither"],["Codex CLI · GPT-6.1 Sol · low (first output, one-line answer)","neither","neither","neither"],["Codex CLI · GPT-6.1 Sol · high (first output, one-line answer)","neither","neither","neither"],["Codex CLI (start-up, first model output)","neither","neither","neither"],["Claude Haiku 4.5 (router, thinking on, via Claude Code)","neither","neither","neither"]]}],["latency-budget-shares","Share of each assumed budget that one step uses (calculation)",{"$k":["key","label","unit"],"$r":[["step","Step","text"],["m300","300 ms: median","text"],["s300","300 ms: slow end","text"],["m800","800 ms: median","text"],["s800","800 ms: slow end","text"],["m1500","1,500 ms: median","text"],["s1500","1,500 ms: slow end","text"]]},{"$k":["step","m300","s300","m800","s800","m1500","s1500"],"$r":[["Deterministic routing policy (Agent, in process)","under 0.1%","under 0.1%","under 0.1%","under 0.1%","under 0.1%","under 0.1%"],["Rule-based System One decision (record write excluded)","0.3%","0.7%","0.1%","0.3%","under 0.1%","0.1%"],["Jev 1.13 (TypeSafe, direct HTTPS)","46%","65%","17%","24%","9.1%","13%"],["OpenAI API · GPT-6 Luna · none (first output, one-line answer)","274%","457%","103%","171%","55%","91%"],["OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)","291%","581%","109%","218%","58%","116%"],["Claude Code · Claude Fable 5.1 (first output, 5 short tasks)","399%","2634%","150%","988%","80%","527%"],["OpenAI API · GPT-6.1 Sol · high (first output, one-line answer)","447%","705%","168%","264%","89%","141%"],["Claude Code · Claude Haiku 4.5 (start-up, first model output)","487%","769%","183%","289%","97%","154%"],["Claude Sonnet 5.5 (router, effort low, via Claude Code)","866%","1433%","325%","537%","173%","287%"],["Codex CLI · GPT-6 Luna · none (first output, one-line answer)","929%","1139%","348%","427%","186%","228%"],["Codex CLI · GPT-6.1 Sol · low (first output, one-line answer)","1251%","1368%","469%","513%","250%","274%"],["Codex CLI · GPT-6.1 Sol · high (first output, one-line answer)","1262%","1432%","473%","537%","252%","286%"],["Codex CLI (start-up, first model output)","1686%","1826%","632%","685%","337%","365%"],["Claude Haiku 4.5 (router, thinking on, via Claude Code)","4225%","11471%","1584%","4302%","845%","2294%"]]}],["latency-budget-in-sequence","How many runs of one step fit back to back in each budget (calculation, counts stop at 1,000)",{"$k":["key","label","unit"],"$r":[["step","Step","text"],["m300","300 ms: at the median","count"],["s300","300 ms: at the slow end","count"],["m800","800 ms: at the median","count"],["s800","800 ms: at the slow end","count"],["m1500","1,500 ms: at the median","count"],["s1500","1,500 ms: at the slow end","count"]]},{"$k":["step","m300","s300","m800","s800","m1500","s1500"],"$r":[["Deterministic routing policy (Agent, in process)",1000,1000,1000,1000,1000,1000],["Rule-based System One decision (record write excluded)",300,150,800,400,1000,750],["Jev 1.13 (TypeSafe, direct HTTPS)",2,1,5,4,10,7],["OpenAI API · GPT-6 Luna · none (first output, one-line answer)",0,0,0,0,1,1],["OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",0,0,0,0,1,0],["Claude Code · Claude Fable 5.1 (first output, 5 short tasks)",0,0,0,0,1,0],["OpenAI API · GPT-6.1 Sol · high (first output, one-line answer)",0,0,0,0,1,0],["Claude Code · Claude Haiku 4.5 (start-up, first model output)",0,0,0,0,1,0],["Claude Sonnet 5.5 (router, effort low, via Claude Code)",0,0,0,0,0,0],["Codex CLI · GPT-6 Luna · none (first output, one-line answer)",0,0,0,0,0,0],["Codex CLI · GPT-6.1 Sol · low (first output, one-line answer)",0,0,0,0,0,0],["Codex CLI · GPT-6.1 Sol · high (first output, one-line answer)",0,0,0,0,0,0],["Codex CLI (start-up, first model output)",0,0,0,0,0,0],["Claude Haiku 4.5 (router, thinking on, via Claude Code)",0,0,0,0,0,0]]}],["latency-budget-router-plus-model","A router in front of a model: the two times added (calculation)",{"$k":["key","label","unit"],"$r":[["router","Router (decision)","text"],["model","Model (first output)","text"],["p50","Sum of medians","ms"],["slow","Sum of slow ends","ms"],["b300","Fits 300 ms","text"],["b800","Fits 800 ms","text"],["b1500","Fits 1,500 ms","text"],["left1500","Left of 1,500 ms at the median (negative: over)","ms"],["leftSlow1500","Left of 1,500 ms at the slow end (negative: over)","ms"]]},{"$k":["router","model","p50","slow","b300","b800","b1500","left1500","leftSlow1500"],"$r":[["Deterministic routing policy (Agent, in process)","OpenAI API · GPT-6 Luna · none (first output, one-line answer)",823,1371,"neither","neither","median and slow end",677,129],["Deterministic routing policy (Agent, in process)","OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",873,1742,"neither","neither","median only",627,-242],["Jev 1.13 (TypeSafe, direct HTTPS)","OpenAI API · GPT-6 Luna · none (first output, one-line answer)",959.5,1566.7,"neither","neither","median only",540.5,-66.7],["Jev 1.13 (TypeSafe, direct HTTPS)","OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",1009.5,1937.7,"neither","neither","median only",490.5,-437.7],["Claude Sonnet 5.5 (router, effort low, via Claude Code)","OpenAI API · GPT-6 Luna · none (first output, one-line answer)",3421,5669,"neither","neither","neither",-1921,-4169],["Claude Sonnet 5.5 (router, effort low, via Claude Code)","OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",3471,6040,"neither","neither","neither",-1971,-4540]]}]]}