{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","range","calculation","polarity"],"$r":[["model-head-to-head","h2h-pass-rate","Pass rate","GPT-6.1 Sol (high) · Codex CLI","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Codex CLI · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","GPT-6.1 Sol (medium) · Codex CLI","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Codex CLI · effort medium · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","GPT-6.1 Sol (low) · Codex CLI","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Codex CLI · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","GPT-6.1 Sol (high) · Codex CLI","h2h-total-latency","Total time per call",5.6,"seconds","5.60 s",15,"\u0001","minmax","Codex CLI · effort high · five short validated tasks",[4.05,19.52],"\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","GPT-6.1 Sol (medium) · Codex CLI","h2h-total-latency","Total time per call",5.65,"seconds","5.65 s",15,"\u0001","minmax","Codex CLI · effort medium · five short validated tasks",[4.1,25.46],"\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","GPT-6.1 Sol (low) · Codex CLI","h2h-total-latency","Total time per call",6.26,"seconds","6.26 s",10,"\u0001","minmax","Codex CLI · effort low · five short validated tasks",[4.65,10.47],"\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","GPT-6.1 Sol (high) · Codex CLI","h2h-first-useful-latency","Time to first useful output",5.32,"seconds","5.32 s",15,"\u0001","minmax","Codex CLI · effort high · five short validated tasks",[3.64,16.37],"\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","GPT-6.1 Sol (medium) · Codex CLI","h2h-first-useful-latency","Time to first useful output",5.05,"seconds","5.05 s",15,"\u0001","minmax","Codex CLI · effort medium · five short validated tasks",[3.36,17.82],"\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","GPT-6.1 Sol (low) · Codex CLI","h2h-first-useful-latency","Time to first useful output",5.14,"seconds","5.14 s",10,"\u0001","minmax","Codex CLI · effort low · five short validated tasks",[4.02,8.5],"\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","GPT-6.1 Sol (high) · Codex CLI","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",6716,"tokens","6,716",15,"\u0001","\u0001","Codex CLI · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","GPT-6.1 Sol (medium) · Codex CLI","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",5180,"tokens","5,180",15,"\u0001","\u0001","Codex CLI · effort medium · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","GPT-6.1 Sol (low) · Codex CLI","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",8064,"tokens","8,064",10,"\u0001","\u0001","Codex CLI · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","GPT-6.1 Sol (high) · Codex CLI","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",5406,"tokens","5,406",15,"\u0001","\u0001","Codex CLI · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","GPT-6.1 Sol (medium) · Codex CLI","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",6943,"tokens","6,943",15,"\u0001","\u0001","Codex CLI · effort medium · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","GPT-6.1 Sol (low) · Codex CLI","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",4059,"tokens","4,059",10,"\u0001","\u0001","Codex CLI · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","GPT-6.1 Sol (high) · Codex CLI","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",42,"tokens","42",15,"\u0001","\u0001","Codex CLI · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",42,"tokens","42",15,"\u0001","\u0001","Codex CLI · effort medium · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","GPT-6.1 Sol (low) · Codex CLI","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",42,"tokens","42",10,"\u0001","\u0001","Codex CLI · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","GPT-6.1 Sol (high) · Codex CLI","h2h-list-price-per-call","List-price cost per call (calculation)",0.01047,"usd","$0.010",15,"\u0001","minmax","Codex CLI · effort high · five short validated tasks",[0.0066,0.02812],true,"\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","GPT-6.1 Sol (medium) · Codex CLI","h2h-list-price-per-call","List-price cost per call (calculation)",0.01018,"usd","$0.010",15,"\u0001","minmax","Codex CLI · effort medium · five short validated tasks",[0.0054,0.02686],true,"\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","GPT-6.1 Sol (low) · Codex CLI","h2h-list-price-per-call","List-price cost per call (calculation)",0.00769,"usd","$0.0077",10,"\u0001","minmax","Codex CLI · effort low · five short validated tasks",[0.00742,0.02649],true,"\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","GPT-6.1 Sol (low) · Codex CLI","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.00998,"usd","$0.010",10,"\u0001","\u0001","Codex CLI · effort low · five short validated tasks","\u0001",true,"\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","GPT-6.1 Sol (high) · Codex CLI","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.01322,"usd","$0.013",15,"\u0001","\u0001","Codex CLI · effort high · five short validated tasks","\u0001",true,"\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","GPT-6.1 Sol (medium) · Codex CLI","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.01564,"usd","$0.016",15,"\u0001","\u0001","Codex CLI · effort medium · five short validated tasks","\u0001",true,"\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort medium · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort high · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort medium · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort high · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",13.11,"seconds","13.1 s",16,"\u0001","minmax","Codex CLI · effort medium · eight hard validated tasks",[8.54,61.6],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",18.12,"seconds","18.1 s",16,"\u0001","minmax","Codex CLI · effort high · eight hard validated tasks",[11.67,92.21],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",10.23,"seconds","10.2 s",16,"\u0001","minmax","Codex CLI · effort medium · eight hard validated tasks",[6.09,40.41],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",12.69,"seconds","12.7 s",16,"\u0001","minmax","Codex CLI · effort high · eight hard validated tasks",[8.93,75.91],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",335,"tokens","335",16,"\u0001","\u0001","Codex CLI · effort medium · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",436,"tokens","436",16,"\u0001","\u0001","Codex CLI · effort high · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.01514,"usd","$0.015",16,"\u0001","\u0001","Codex CLI · effort high · eight hard validated tasks","\u0001",true,"\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.02564,"usd","$0.026",16,"\u0001","\u0001","Codex CLI · effort medium · eight hard validated tasks","\u0001",true,"\u0001"],["coding-agents-head-to-head","coding-agents-pass-rate","Passed every hidden check","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-pass-rate","Coding sessions that passed every hidden check",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests","\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-wall-time","Wall time per session","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-wall-time","Time per coding session",113.4,"seconds","113.4 s",12,"\u0001","minmax","Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",[78.5,221.9],"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-tool-calls","Tool calls per session","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-tool-calls","Tool calls per coding session",12.5,"calls","12.5",12,"\u0001","minmax","Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",[8,18],"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-cost-per-pass","List-price cost per pass","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-cost-per-pass","List-price cost per passing coding session (calculation)",0.0978,"usd","$0.098",12,"\u0001","\u0001","Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests","\u0001",true,"\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-total-latency","Total time per call by effort on hard tasks",13.62,"seconds","13.6 s",16,"\u0001","minmax","Codex CLI · effort low · eight hard validated tasks, effort ladder",[7.94,44.29],"\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-total-latency","Total time per call by effort on hard tasks",13.11,"seconds","13.1 s",16,"\u0001","minmax","Codex CLI · effort medium · eight hard validated tasks, effort ladder",[8.54,61.6],"\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-total-latency","Total time per call by effort on hard tasks",18.12,"seconds","18.1 s",16,"\u0001","minmax","Codex CLI · effort high · eight hard validated tasks, effort ladder",[11.67,92.21],"\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",284,"tokens","284",16,"\u0001","\u0001","Codex CLI · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",335,"tokens","335",16,"\u0001","\u0001","Codex CLI · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",436,"tokens","436",16,"\u0001","\u0001","Codex CLI · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01284,"usd","$0.013",16,"\u0001","\u0001","Codex CLI · effort low · eight hard validated tasks, effort ladder","\u0001",true,"\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.02564,"usd","$0.026",16,"\u0001","\u0001","Codex CLI · effort medium · eight hard validated tasks, effort ladder","\u0001",true,"\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01514,"usd","$0.015",16,"\u0001","\u0001","Codex CLI · effort high · eight hard validated tasks, effort ladder","\u0001",true,"\u0001"],["caching-consistency","consistency-pass-rate","Exact number","GPT-6.1 Sol (medium) · Codex CLI","consistency-pass-rate/Exact number","Same prompt, 10 times: strict pass rate (Exact number)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","JSON object","GPT-6.1 Sol (medium) · Codex CLI","consistency-pass-rate/JSON object","Same prompt, 10 times: strict pass rate (JSON object)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Code fix","GPT-6.1 Sol (medium) · Codex CLI","consistency-pass-rate/Code fix","Same prompt, 10 times: strict pass rate (Code fix)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Exact number","GPT-6.1 Sol (medium) · Codex CLI","consistency-distinct-answers/Exact number","Same prompt, 10 times: how many different answers (Exact number)",1,"count","1",10,"\u0001","\u0001","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","JSON object","GPT-6.1 Sol (medium) · Codex CLI","consistency-distinct-answers/JSON object","Same prompt, 10 times: how many different answers (JSON object)",1,"count","1",10,"\u0001","\u0001","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Code fix","GPT-6.1 Sol (medium) · Codex CLI","consistency-distinct-answers/Code fix","Same prompt, 10 times: how many different answers (Code fix)",6,"count","6",10,"\u0001","\u0001","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Exact number","GPT-6.1 Sol (medium) · Codex CLI","consistency-latency-spread/Exact number","Same prompt, 10 times: time per call (Exact number)",13.38,"seconds","13.4 s",10,"\u0001","minmax","Codex CLI · effort medium · same prompt repeated 10 times",[12.29,17.97],"\u0001","\u0001"],["caching-consistency","consistency-latency-spread","JSON object","GPT-6.1 Sol (medium) · Codex CLI","consistency-latency-spread/JSON object","Same prompt, 10 times: time per call (JSON object)",6.42,"seconds","6.42 s",10,"\u0001","minmax","Codex CLI · effort medium · same prompt repeated 10 times",[5.25,8.26],"\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Code fix","GPT-6.1 Sol (medium) · Codex CLI","consistency-latency-spread/Code fix","Same prompt, 10 times: time per call (Code fix)",11.29,"seconds","11.3 s",10,"\u0001","minmax","Codex CLI · effort medium · same prompt repeated 10 times",[9.08,14.85],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","Total time","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-exact-reply-latency/Total time","CLI vs API: time for a one-line answer (Total time)",4.18,"seconds","4.18 s",5,"\u0001","minmax","Codex CLI · effort low · fixed exact reply, 5 runs",[3.86,4.53],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","Total time","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-exact-reply-latency/Total time","CLI vs API: time for a one-line answer (Total time)",4.19,"seconds","4.19 s",5,"\u0001","minmax","Codex CLI · effort high · fixed exact reply, 5 runs",[3.81,4.69],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","First useful output","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-exact-reply-latency/First useful output","CLI vs API: time for a one-line answer (First useful output)",3.75,"seconds","3.75 s",5,"\u0001","minmax","Codex CLI · effort low · fixed exact reply, 5 runs",[3.44,4.1],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","First useful output","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-exact-reply-latency/First useful output","CLI vs API: time for a one-line answer (First useful output)",3.79,"seconds","3.79 s",5,"\u0001","minmax","Codex CLI · effort high · fixed exact reply, 5 runs",[3.37,4.3],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","Total time","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-small-coding-latency/Total time","CLI vs API: time for a small coding task (Total time)",14.15,"seconds","14.2 s",3,"\u0001","minmax","Codex CLI · effort low · small coding task, 3 runs",[13.02,14.41],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","Total time","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-small-coding-latency/Total time","CLI vs API: time for a small coding task (Total time)",17.85,"seconds","17.9 s",3,"\u0001","minmax","Codex CLI · effort high · small coding task, 3 runs",[17.68,22.42],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","First useful output","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-small-coding-latency/First useful output","CLI vs API: time for a small coding task (First useful output)",13.6,"seconds","13.6 s",3,"\u0001","minmax","Codex CLI · effort low · small coding task, 3 runs",[12.52,13.83],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","First useful output","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-small-coding-latency/First useful output","CLI vs API: time for a small coding task (First useful output)",17.27,"seconds","17.3 s",3,"\u0001","minmax","Codex CLI · effort high · small coding task, 3 runs",[17.13,21.86],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-prompt-overhead","Input tokens","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-prompt-overhead","Hidden prompt: input tokens for the same one-line request",19551,"tokens","19,551",5,"\u0001","\u0001","Codex CLI · effort low · short fixed tasks","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-prompt-overhead","Input tokens","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-prompt-overhead","Hidden prompt: input tokens for the same one-line request",19555,"tokens","19,555",5,"\u0001","\u0001","Codex CLI · effort high · short fixed tasks","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","Total time","Codex CLI · GPT-6.1 Sol · medium","scheduler-repair-claude-vs-codex/Total time","Repairing a scheduler: Claude Code vs Codex vs API (Total time)",61.16,"seconds","61.2 s",3,"\u0001","minmax","Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs",[59.9,69.51],"\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","First useful output","Codex CLI · GPT-6.1 Sol · medium","scheduler-repair-claude-vs-codex/First useful output","Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",15.56,"seconds","15.6 s",3,"\u0001","minmax","Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs",[13.65,23.04],"\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-output-tokens","Output tokens","Codex CLI · GPT-6.1 Sol · medium","scheduler-repair-output-tokens/Output tokens","Output tokens to repair the scheduler (Output tokens)",1181,"tokens","1,181",3,"\u0001","\u0001","Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Codex CLI · effort low · instructions","\u0001",true,"higher"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Codex CLI · effort low · JSON schema","\u0001",true,"higher"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Codex CLI · effort low · instructions","\u0001",true,"higher"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Codex CLI · effort low · JSON schema","\u0001",true,"higher"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","Codex CLI · effort low · instructions","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","Codex CLI · effort low · JSON schema","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · instructions","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · JSON schema","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · instructions","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · JSON schema","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · instructions","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · JSON schema","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-time","Time per call, instructions vs schema mode",6.21,"seconds","6.21 s",12,"\u0001","minmax","Codex CLI · effort low · instructions",[4.2,12.27],"\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-time","Time per call, instructions vs schema mode",5.96,"seconds","5.96 s",12,"\u0001","minmax","Codex CLI · effort low · JSON schema",[4.62,20.97],"\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",117,"tokens","117",12,"\u0001","\u0001","Codex CLI · effort low · instructions","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",123,"tokens","123",12,"\u0001","\u0001","Codex CLI · effort low · JSON schema","\u0001","\u0001","\u0001"],["thinking-token-bill","thinking-bill-share","Median call","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",57.01,"percent","57%",16,"\u0001","minmax","Codex CLI · effort high",[29.19,90.8],true,"none"],["thinking-token-bill","thinking-bill-share","Median call","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",46.33,"percent","46.3%",16,"\u0001","minmax","Codex CLI · effort medium",[11.42,86.85],true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.002273,"usd","$0.0023",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.004114,"usd","$0.0041",16,"\u0001","\u0001","Codex CLI · effort high","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.003025,"usd","$0.0030",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.002905,"usd","$0.0029",16,"\u0001","\u0001","Codex CLI · effort high","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.020339,"usd","$0.020",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.008117,"usd","$0.0081",16,"\u0001","\u0001","Codex CLI · effort high","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","GPT-6.1 Sol (low) · Codex CLI","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.001223,"usd","$0.0012",16,"\u0001","\u0001","Codex CLI · effort low","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.002273,"usd","$0.0023",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.004114,"usd","$0.0041",16,"\u0001","\u0001","Codex CLI · effort high","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","GPT-6.1 Sol (low) · Codex CLI","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.012837,"usd","$0.013",16,"\u0001","\u0001","Codex CLI · effort low","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.025637,"usd","$0.026",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.015137,"usd","$0.015",16,"\u0001","\u0001","Codex CLI · effort high","\u0001",true,"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",46.33,"percent","46.3%",16,"\u0001","minmax","Codex CLI · effort medium",[11.42,86.85],true,"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",57.01,"percent","57%",16,"\u0001","minmax","Codex CLI · effort high",[29.19,90.8],true,"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",41.05,"percent","41%",15,"\u0001","minmax","Codex CLI · effort medium",[0,71.43],true,"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",58.06,"percent","58.1%",15,"\u0001","minmax","Codex CLI · effort high",[0,75.76],true,"none"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",3.52,"seconds","3.52 s",4,"\u0001","minmax","Codex CLI · effort low",[2.75,4.42],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",79.6,"tokens","80",4,"\u0001","minmax","Codex CLI · effort low",[71.6,80.5],true,"\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",323,"count","323",4,"\u0001","minmax","Codex CLI · effort low",[291,327],true,"\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","GPT-6.1 Sol (low) · Codex CLI","1k","speed-anatomy-prompt-size/1k","Time to first text as the prompt grows: 1k",3.36,"seconds","3.36 s",3,"\u0001","minmax","Codex CLI · effort low",[3.36,4.75],true,"\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","GPT-6.1 Sol (low) · Codex CLI","16k","speed-anatomy-prompt-size/16k","Time to first text as the prompt grows: 16k",4.02,"seconds","4.02 s",3,"\u0001","minmax","Codex CLI · effort low",[3.3,4.28],true,"\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","GPT-6.1 Sol (low) · Codex CLI","64k","speed-anatomy-prompt-size/64k","Time to first text as the prompt grows: 64k",3.93,"seconds","3.93 s",3,"\u0001","minmax","Codex CLI · effort low",[3.42,4.38],true,"\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","1k prompt","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-total-by-size/1k prompt","Total time per call by prompt size (1k prompt)",3.43,"seconds","3.43 s",3,"\u0001","minmax","Codex CLI · effort low",[3.43,4.92],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","16k prompt","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-total-by-size/16k prompt","Total time per call by prompt size (16k prompt)",4.14,"seconds","4.14 s",3,"\u0001","minmax","Codex CLI · effort low",[3.96,4.68],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","64k prompt","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-total-by-size/64k prompt","Total time per call by prompt size (64k prompt)",3.96,"seconds","3.96 s",3,"\u0001","minmax","Codex CLI · effort low",[3.47,4.44],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-lookup-correct","Exact answer","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-lookup-correct","Exact lookup answers at the 1k, 16k and 64k prompt-size targets",1,"rate","100% (9/9)",9,[0.7009,1],"ci95","Codex CLI · effort low","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Strict pass","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-pass-rate/Strict pass","Pass rate on 4 harder tasks (Strict pass)",0.6875,"rate","69% (11/16)",16,[0.444,0.8584],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Lenient (format misses counted)","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-pass-rate/Lenient (format misses counted)","Pass rate on 4 harder tasks (Lenient (format misses counted))",0.6875,"rate","69% (11/16)",16,[0.444,0.8584],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-tool-attempts","Tool attempt","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-tool-attempts","Calls that tried a tool although tools were off",0,"rate","0% (0/16)",16,[0,0.1936],"ci95","Codex CLI · effort medium","\u0001","\u0001","none"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","10x10 nonogram","harder-h2h-pass-by-task/10x10 nonogram","Strict pass rate by task: 10x10 nonogram",0.75,"rate","75% (3/4)",4,[0.3006,0.9544],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","Sudoku, 22 givens","harder-h2h-pass-by-task/Sudoku, 22 givens","Strict pass rate by task: Sudoku, 22 givens",0.25,"rate","25% (1/4)",4,[0.0456,0.6994],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","6x6 Skyscrapers","harder-h2h-pass-by-task/6x6 Skyscrapers","Strict pass rate by task: 6x6 Skyscrapers",1,"rate","100% (4/4)",4,[0.5101,1],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","Seeded shuffle output","harder-h2h-pass-by-task/Seeded shuffle output","Strict pass rate by task: Seeded shuffle output",0.75,"rate","75% (3/4)",4,[0.3006,0.9544],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-total-latency","Total time per call","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-total-latency","Total time per call on harder tasks",120.24,"seconds","120.2 s",13,"\u0001","minmax","Codex CLI · effort medium",[46.24,273.46],"\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-output-tokens/Output tokens","Output tokens per call on harder tasks (Output tokens)",4994,"tokens","4,994",13,"\u0001","minmax","Codex CLI · effort medium",[2099,13413],"\u0001","none"],["harder-tasks-head-to-head","harder-h2h-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-cost-per-pass","List-price cost per strict pass on harder tasks (calculation)",0.08293,"usd","$0.083",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"\u0001"]]}