{"$k":["id","title","columns","rows"],"$r":[["routing-head-to-head","Same cases, two routers (Jev: its recorded production run, one pass)",{"$k":["key","label","unit"],"$r":[["pair","Pair","text"],["bothRight","Both right","count"],["onlyA","Only first right","count"],["onlyB","Only second right","count"],["bothWrong","Both wrong","count"],["p","Exact McNemar p","\u0001"]]},[{"pair":"Claude Haiku 4.5 vs Jev 1.13 (TypeSafe)","bothRight":70,"onlyA":3,"onlyB":4,"bothWrong":5,"p":1},{"pair":"Claude Sonnet 5.5 vs Jev 1.13 (TypeSafe)","bothRight":73,"onlyA":4,"onlyB":1,"bothWrong":4,"p":0.375}]],["routing-question-accuracy","Per-question accuracy by decision type (Jev: its recorded production run, one pass)",{"$k":["key","label","unit"],"$r":[["decision","Decision","text"],["question","Question","text"],["jev","Jev 1.13 (TypeSafe)","text"],["claude-haiku","Claude Haiku 4.5","text"],["claude-sonnet","Claude Sonnet 5.5","text"]]},{"$k":["decision","question","jev","claude-haiku","claude-sonnet"],"$r":[["Failure class","failure","18/18","17/18","18/18"],["Message intent","intent","20/20","20/20","20/20"],["Is it a rule?","kind","12/12","12/12","12/12"],["Context shape","turn","5/6","5/6","5/6"],["Context shape","transcript","12/16","13/16","15/16"],["Context shape","artifacts","4/4","4/4","4/4"],["Context shape","knowledge","22/24","22/24","23/24"],["Context shape","memories","21/21","21/21","20/21"],["Context shape","examples","10/10","10/10","10/10"],["Context shape","scope","30/31","29/31","31/31"],["Context shape","complexity","31/32","30/32","31/32"]]}],["routing-routers","Every router",{"$k":["key","label","unit"],"$r":[["router","Router","text"],["how","How measured","text"],["exact","Exact","text"],["keys","Key accuracy","text"],["tokens","Tokens per decision (input incl. cache / output)","text"],["cost","USD per 1,000","usd"]]},{"$k":["router","how","exact","keys","tokens","cost"],"$r":[["Jev 1.13 (TypeSafe)","live API run 2026-10-06: 3 repeats of the same 82 decisions, 246 counted calls one at a time from one Mac. All calls: 221 of 246 exact and 552 of 582 questions; the counts shown are on the 82-decision scale of the interval. The recorded production run of 2026-10-05 scored 74 of 82. Cost is a calculation from the reported input tokens","74/82","184/194","803 / 148",0.0337],["Claude Haiku 4.5","this benchmark, Claude Code CLI on a subscription account, one call per decision","73/82","183/194","1,831 / 1,419",8.924],["Claude Sonnet 5.5","this benchmark, Claude Code CLI on a subscription account, one call per decision","77/82","189/194","1,785 / 107",4.996],["Clef / Clef-Flash (local)","not measured: No local Clef server was running and installing a 6-20 GB model was out of scope for this run.","not measured","not measured","",null]]}],["routing-jev-live-repeats","Jev live run, repeat by repeat",{"$k":["key","label","unit"],"$r":[["run","Run","text"],["exact","Exact decisions","text"],["keys","Questions answered acceptably","text"],["medianMs","Median time per call (ms)","ms"]]},{"$k":["run","exact","keys","medianMs"],"$r":[["Repeat 1 (without the cold first call)","74/82","184/194",130.3],["Repeat 2","73/82","183/194",141.7],["Repeat 3","74/82","185/194",137.1],["All 3 repeats (246 calls)","221/246 = 89.8%; case-level 95% interval 81.9% to 95.0%","552/582 = 94.8%; 90.8% to 97.2%",136.5],["Recorded production run, 2026-10-05 (one pass, no latency recorded)","74/82","185/194",null]]}]]}