{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","range","polarity","calculation","statId"],"$r":[["system-one-arena","arena-accuracy","Accuracy","Jev 1.13","arena-accuracy","Who decides right? Accuracy on 1,000+ checkable decisions",0.7681,"rate","77% (805/1048)",1048,[0.7416,0.7927],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-suite","Jev 1.13","Games","arena-by-suite/Games","Where each model is strong: Games",0.4219,"rate","42% (81/192)",192,[0.3542,0.4926],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-suite","Jev 1.13","Logic and thought experiments","arena-by-suite/Logic and thought experiments","Where each model is strong: Logic and thought experiments",0.7436,"rate","74% (116/156)",156,[0.6698,0.8057],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-suite","Jev 1.13","Policy cases, 20 industries","arena-by-suite/Policy cases, 20 industries","Where each model is strong: Policy cases, 20 industries",0.8139,"rate","81% (223/274)",274,[0.7636,0.8555],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-suite","Jev 1.13","Usability intents","arena-by-suite/Usability intents","Where each model is strong: Usability intents",0.9336,"rate","93% (211/226)",226,[0.8934,0.9594],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-suite","Jev 1.13","Stress tests","arena-by-suite/Stress tests","Where each model is strong: Stress tests",0.87,"rate","87% (174/200)",200,[0.8163,0.9097],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-latency-same-gpu","Hosted: network included","Jev 1.13","arena-latency-same-gpu/Hosted: network included","Speed on one GPU: third-party numbers (Hosted: network included)",524.1,"ms","524 ms","\u0001","\u0001","p50-p95","",[524.1,536],"lower","\u0001","\u0001"],["system-one-arena","arena-latency-gateway","Median latency, one gateway","Jev 1.13 (TypeSafe)","arena-latency-gateway","Speed through one gateway: OpenRouter's own numbers",0.17,"seconds","0.17 s","\u0001","\u0001","\u0001","","\u0001","lower","\u0001","\u0001"],["system-one-arena","arena-cost-same-provider","USD per 1,000 decisions","Jev 1.13 ($0.042/M, 825 tokens)","arena-cost-same-provider","Price per 1,000 decisions at one provider's list prices",0.0347,"usd","$0.035","\u0001","\u0001","\u0001","$0.042/M · 825 tokens","\u0001","lower",true,"\u0001"],["system-one-arena","arena-robust","First presentation","Jev 1.13","arena-robust/First presentation","Same question, different presentation (First presentation)",0.7681,"rate","77% (805/1048)",1048,[0.7416,0.7927],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-robust","All three presentations","Jev 1.13","arena-robust/All three presentations","Same question, different presentation (All three presentations)",0.729,"rate","73% (764/1048)",1048,[0.7013,0.755],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-flips","Options shuffled","Jev 1.13","arena-flips/Options shuffled","Decisions that changed when only the presentation changed (Options shuffled)",0.077,"rate","8% (74/961)",961,[0.0618,0.0956],"ci95","","\u0001","lower","\u0001","\u0001"],["system-one-arena","arena-flips","Options renamed a, b, c","Jev 1.13","arena-flips/Options renamed a, b, c","Decisions that changed when only the presentation changed (Options renamed a, b, c)",0.0687,"rate","7% (66/961)",961,[0.0543,0.0864],"ci95","","\u0001","lower","\u0001","\u0001"],["system-one-arena","arena-escape","Escaped when it should (higher is better)","Jev 1.13","arena-escape/Escaped when it should (higher is better)","Knowing when to say \"none of these\" (Escaped when it should (higher is better))",0.8571,"rate","86% (126/147)",147,[0.7915,0.9046],"ci95","","\u0001","none","\u0001","\u0001"],["system-one-arena","arena-escape","Escaped when it should not (lower is better)","Jev 1.13","arena-escape/Escaped when it should not (lower is better)","Knowing when to say \"none of these\" (Escaped when it should not (lower is better))",0.055,"rate","6% (34/618)",618,[0.0396,0.0759],"ci95","","\u0001","none","\u0001","\u0001"],["system-one-arena","arena-injection","Right despite the injected text","Jev 1.13","arena-injection","Prompt injection: does text in the state hijack the decision?",0.95,"rate","95% (38/40)",40,[0.835,0.9862],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-length","Up to 512 tokens","Jev 1.13","arena-by-length/Up to 512 tokens","Short inputs vs long inputs (Up to 512 tokens)",0.7867,"rate","79% (177/225)",225,[0.7286,0.8351],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-length","More than 512 tokens","Jev 1.13","arena-by-length/More than 512 tokens","Short inputs vs long inputs (More than 512 tokens)",0.7631,"rate","76% (628/823)",823,[0.7328,0.7908],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-confident-wrong","Confident wrong answers","Jev 1.13","arena-confident-wrong","Wrong and sure of it",0.1111,"rate","11% (27/243)",243,[0.0775,0.1568],"ci95","","\u0001","lower","\u0001","\u0001"],["system-one-arena","arena-elo","Elo","Jev 1.13","arena-elo","Tournament rating across every game",1040,"score","1040.00",336,"\u0001","minmax","",[1002,1082],"\u0001","\u0001","\u0001"],["system-one-arena","arena-wdl","Wins","Jev 1.13","arena-wdl/Wins","Wins, draws and losses in the round robin (Wins)",196,"count","196",336,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-wdl","Draws","Jev 1.13","arena-wdl/Draws","Wins, draws and losses in the round robin (Draws)",6,"count","6",336,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-wdl","Losses","Jev 1.13","arena-wdl/Losses","Wins, draws and losses in the round robin (Losses)",134,"count","134",336,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-perfect-moves","Jev 1.13","Tic-tac-toe","arena-perfect-moves/Tic-tac-toe","How often a model found the perfect move: Tic-tac-toe",0.4653,"rate","47% (67/144)",144,[0.3858,0.5466],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-perfect-moves","Jev 1.13","Connect Four","arena-perfect-moves/Connect Four","How often a model found the perfect move: Connect Four",0.3789,"rate","38% (133/351)",351,[0.3297,0.4307],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-perfect-moves","Jev 1.13","Nim","arena-perfect-moves/Nim","How often a model found the perfect move: Nim",0.3916,"rate","39% (65/166)",166,[0.3206,0.4675],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-perfect-moves","Jev 1.13","Dots and Boxes","arena-perfect-moves/Dots and Boxes","How often a model found the perfect move: Dots and Boxes",0.5479,"rate","55% (423/772)",772,[0.5127,0.5827],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-pong","Pong score","Jev 1.13","arena-pong","Pong as deployed: who won",0.8125,"rate","81% (13/16)",16,[0.5699,0.9341],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-pong-quality","Right zone","Jev 1.13","arena-pong-quality","Pong decision quality, ignoring time",1,"rate","100% (523/523)",523,[0.9927,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","\u0001","\u0001","\u0001","stat:arena-jev-latency","Jev 1.13 hosted API: median call time from Houston, network included (not comparable with a model on another machine)",137,"ms","137 ms",120,"\u0001","\u0001","","\u0001","\u0001","\u0001","arena-jev-latency"],["routing-jev-vs-llm","routing-exact-decisions","Exact rate","Jev 1.13 (TypeSafe)","routing-exact-decisions","Typed routing decisions answered exactly right",0.8984,"rate","90%",82,[0.8191,0.9497],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-key-accuracy","Key accuracy","Jev 1.13 (TypeSafe)","routing-key-accuracy","Per-question accuracy",0.9485,"rate","95% (184/194)",194,[0.9077,0.9718],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Jev 1.13 (TypeSafe)","Failure class","routing-exact-by-decision/Failure class","Exact rate by decision type: Failure class",1,"rate","100% (18/18)",18,[0.8241,1],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Jev 1.13 (TypeSafe)","Message intent","routing-exact-by-decision/Message intent","Exact rate by decision type: Message intent",1,"rate","100% (20/20)",20,[0.8389,1],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Jev 1.13 (TypeSafe)","Is it a rule?","routing-exact-by-decision/Is it a rule?","Exact rate by decision type: Is it a rule?",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Jev 1.13 (TypeSafe)","Context shape","routing-exact-by-decision/Context shape","Exact rate by decision type: Context shape",0.7396,"rate","74%",32,[0.5789,0.8675],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-cost-per-1000","Cost","Jev 1.13 (TypeSafe)","routing-cost-per-1000","Cost per 1,000 routing decisions",0.0337,"usd","$0.034",246,"\u0001","\u0001","typed routing decisions · TypeSafe API","\u0001","\u0001",true,"\u0001"],["routing-jev-vs-llm","routing-decision-latency","Wall time (direct API call)","Jev 1.13 (TypeSafe)","routing-decision-latency/Wall time (direct API call)","Time per routing decision (Wall time (direct API call))",136.5,"ms","137 ms",246,"\u0001","p50-p95","typed routing decisions · TypeSafe API",[136.5,195.7],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-decision-latency","Decision time","Jev 1.13 (TypeSafe)","router-overhead-decision-latency","Time to make one routing decision",136.5,"ms","137 ms",246,"\u0001","p50-p95","routing overhead per decision · TypeSafe API",[136.5,195.7],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-completed","Completed","Jev 1.13 (TypeSafe)","router-overhead-completed","Routing calls that returned a decision",1,"rate","100% (246/246)",246,[0.9846,1],"ci95","routing overhead per decision · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-reported","Cost per 1,000 decisions","Jev 1.13 (TypeSafe)","router-overhead-cost-reported","Cost per 1,000 routing decisions: no model call vs provider-reported",0.0337,"usd","$0.034",82,"\u0001","\u0001","routing overhead per decision · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-list-price","Cost per 1,000 decisions (list price)","Jev 1.13 (TypeSafe)","router-overhead-cost-list-price","Cost per 1,000 routing decisions for the model routers (calculation)",0.0337,"usd","$0.034",246,"\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price · routing overhead per decision · TypeSafe API","\u0001","\u0001",true,"\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Every model call routed (49.5 per task)","Jev 1.13 (TypeSafe)","router-overhead-cost-per-1000-tasks/Every model call routed (49.5 per task)","Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",1.67,"usd","$1.67","\u0001","\u0001","\u0001","calculation per 1,000 tasks from recorded decision counts · TypeSafe API","\u0001","\u0001",true,"\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Only System One decisions (7 per task)","Jev 1.13 (TypeSafe)","router-overhead-cost-per-1000-tasks/Only System One decisions (7 per task)","Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",0.24,"usd","$0.24","\u0001","\u0001","\u0001","calculation per 1,000 tasks from recorded decision counts · TypeSafe API","\u0001","\u0001",true,"\u0001"],["routing-overhead","router-overhead-delay-per-task","Every model call routed (49.5 per task)","Jev 1.13 (TypeSafe)","router-overhead-delay-per-task/Every model call routed (49.5 per task)","Added routing delay per task (calculation) (Every model call routed (49.5 per task))",6.7568,"seconds","6.76 s","\u0001","\u0001","\u0001","calculation per task from recorded decision counts, decisions in line · TypeSafe API","\u0001","\u0001",true,"\u0001"],["routing-overhead","router-overhead-delay-per-task","Only System One decisions (7 per task)","Jev 1.13 (TypeSafe)","router-overhead-delay-per-task/Only System One decisions (7 per task)","Added routing delay per task (calculation) (Only System One decisions (7 per task))",0.9555,"seconds","0.96 s","\u0001","\u0001","\u0001","calculation per task from recorded decision counts, decisions in line · TypeSafe API","\u0001","\u0001",true,"\u0001"],["cost-thought-experiments","repriced-cost-per-resolved","Repriced cost per resolved instance","Jev 1.13 (router)","repriced-cost-per-resolved","Thought experiment: the same tokens at other list prices",0.042,"usd","$0.042","\u0001","\u0001","\u0001","router · calculation: Agent’s recorded tokens at this model’s list price","\u0001","\u0001",true,"\u0001"],["routing-holdout","routing-holdout-exact","Exact rate","Jev 1.13 (TypeSafe)","routing-holdout-exact","Unseen routing decisions answered exactly right",0.8214,"rate","82% (46/56)",56,[0.7016,0.9],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-key-accuracy","Key accuracy","Jev 1.13 (TypeSafe)","routing-holdout-key-accuracy","Per-question accuracy on unseen decisions",0.904,"rate","90% (113/125)",125,[0.8397,0.9442],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-by-purpose","Jev 1.13 (TypeSafe)","Failure class","routing-holdout-by-purpose/Failure class","Exact rate on unseen decisions, by decision type: Failure class",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-by-purpose","Jev 1.13 (TypeSafe)","Message intent","routing-holdout-by-purpose/Message intent","Exact rate on unseen decisions, by decision type: Message intent",0.8571,"rate","86% (12/14)",14,[0.6006,0.9599],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-by-purpose","Jev 1.13 (TypeSafe)","Is it a rule?","routing-holdout-by-purpose/Is it a rule?","Exact rate on unseen decisions, by decision type: Is it a rule?",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-by-purpose","Jev 1.13 (TypeSafe)","Context shape","routing-holdout-by-purpose/Context shape","Exact rate on unseen decisions, by decision type: Context shape",0.5714,"rate","57% (8/14)",14,[0.3259,0.7862],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Tuned set (routing-jev-vs-llm)","Jev 1.13 (TypeSafe)","routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)","Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",0.9024,"rate","90% (74/82)",82,[0.8191,0.9497],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Unseen holdout","Jev 1.13 (TypeSafe)","routing-holdout-tuned-vs-unseen/Unseen holdout","Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",0.8214,"rate","82% (46/56)",56,[0.7016,0.9],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-latency","Wall time","Jev 1.13 (TypeSafe)","routing-holdout-latency/Wall time","Time per routing decision, by route (Wall time)",0.139,"seconds","0.14 s",168,"\u0001","p50-p95","",[0.139,0.192],"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-cost-per-1000","Cost","Jev 1.13 (TypeSafe)","routing-holdout-cost-per-1000","Cost per 1,000 unseen routing decisions",0.03065,"usd","$0.031",168,"\u0001","\u0001","","\u0001","\u0001",true,"\u0001"],["routing-holdout","\u0001","\u0001","\u0001","stat:holdout-jev-stability","Jev 1.13 (TypeSafe): same answers on every key in 3 repetitions",0.9464,"rate","95% (53/56)",56,[0.8539,0.9816],"ci95","TypeSafe","\u0001","\u0001","\u0001","holdout-jev-stability"],["routing-holdout","\u0001","\u0001","\u0001","stat:holdout-gap-jev","Jev 1.13 (TypeSafe): holdout minus tuned-set exact rate",-0.081,"rate","−8.1 points",56,"\u0001","\u0001","TypeSafe","\u0001","\u0001",true,"holdout-gap-jev"]]}