[{"slug":"inference-provider-index","chart":{"id":"provider-prices-kimi-k3","title":"Kimi K3: price per million tokens by provider","subtitle":"Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06","kind":"grouped-bar","unit":"usd","yLabel":"USD per million tokens","series":{"$k":["name","points"],"$r":[["Input",{"$k":["label","value"],"$r":[["Relace (fp4)",0.83],["Phala",1.95],["Sail Research (fp4)",0.84],["Decart (mxfp4)",2.01],["InferenceNet (fp4)",0.95],["Wafer",0.95],["Morph (fp8)",1.274],["Makora",1.53],["AkashML (fp4)",1.3],["DigitalOcean",2.55],["Together",2.7],["DeepInfra (mxfp4)",2.85],["BaseTen (fp8)",3],["Chutes (mxfp4)",3],["Fireworks",3],["Modal (mxfp4)",3],["Moonshot AI (mxfp4)",3],["Parasail (fp4)",3],["Alibaba",3.45]]}],["Output",{"$k":["label","value"],"$r":[["Relace (fp4)",13],["Phala",9.75],["Sail Research (fp4)",13.5],["Decart (mxfp4)",10.05],["InferenceNet (fp4)",14],["Wafer",14],["Morph (fp8)",13.296],["Makora",12.75],["AkashML (fp4)",14],["DigitalOcean",12.95],["Together",13.5],["DeepInfra (mxfp4)",14.25],["BaseTen (fp8)",15],["Chutes (mxfp4)",15],["Fireworks",15],["Modal (mxfp4)",15],["Moonshot AI (mxfp4)",15],["Parasail (fp4)",15],["Alibaba",17.25]]}],["Cache read",{"$k":["label","value"],"$r":[["Relace (fp4)",0.45],["Phala",0.195],["Sail Research (fp4)",0.3],["Decart (mxfp4)",0.201],["InferenceNet (fp4)",0.31],["Wafer",0.4],["Morph (fp8)",0.278],["Makora",0.204],["AkashML (fp4)",1.3],["DigitalOcean",0.255],["Together",0.27],["DeepInfra (mxfp4)",0.285],["BaseTen (fp8)",0.3],["Chutes (mxfp4)",0.3],["Fireworks",0.3],["Modal (mxfp4)",0.3],["Moonshot AI (mxfp4)",0.3],["Parasail (fp4)",0.3],["Alibaba",0.345]]}]]},"note":"Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.","factContext":"Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06","sourceIds":["openrouter-api-snapshot"]}},{"slug":"inference-provider-index","chart":{"id":"provider-prices-glm-5-3","title":"GLM 5.3: price per million tokens by provider","subtitle":"Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06","kind":"grouped-bar","unit":"usd","yLabel":"USD per million tokens","series":{"$k":["name","points"],"$r":[["Input",{"$k":["label","value"],"$r":[["Novita (fp8)",0.42],["Reka",0.17],["Sail Research (fp8)",0.2],["Morph (fp8)",0.179],["DeepInfra (fp4)",0.5625],["SiliconFlow (fp8)",0.7],["InferenceNet",0.14],["Makora (fp4)",0.18],["AkashML (fp8)",0.19],["Phala",0.84],["Inceptron (fp4)",0.6],["DigitalOcean",0.91],["GMICloud (fp8)",0.98],["Alibaba",1.19],["Decart (fp4)",1.19],["Wafer",0.15],["Friendli",1.26],["AtlasCloud (fp8)",1.4],["Baidu (fp8)",1.4],["BaseTen (fp4)",1.4],["Cloudflare",1.4],["Crusoe (fp4)",1.4],["Fireworks",1.4],["Mistral (nvfp4)",1.4],["Modal",1.4],["Nebius (fp4)",1.4],["Parasail (fp8)",1.4],["PrimeIntellect",1.4],["Together",1.4],["Venice",1.4],["Z.AI (fp8)",1.4],["Relace",0.03]]}],["Output",{"$k":["label","value"],"$r":[["Novita (fp8)",1.32],["Reka",3],["Sail Research (fp8)",3.4],["Morph (fp8)",3.553],["DeepInfra (fp4)",2.5],["SiliconFlow (fp8)",2.2],["InferenceNet",4.4],["Makora (fp4)",4.4],["AkashML (fp8)",4.4],["Phala",2.64],["Inceptron (fp4)",3.39],["DigitalOcean",2.86],["GMICloud (fp8)",3.08],["Alibaba",3.74],["Decart (fp4)",3.74],["Wafer",7],["Friendli",3.96],["AtlasCloud (fp8)",4.4],["Baidu (fp8)",4.4],["BaseTen (fp4)",4.4],["Cloudflare",4.4],["Crusoe (fp4)",4.4],["Fireworks",4.4],["Mistral (nvfp4)",4.4],["Modal",4.4],["Nebius (fp4)",4.4],["Parasail (fp8)",4.4],["PrimeIntellect",4.4],["Together",4.4],["Venice",4.4],["Z.AI (fp8)",4.4],["Relace",12]]}],["Cache read",{"$k":["label","value"],"$r":[["Novita (fp8)",0.078],["Reka",0.169],["Sail Research (fp8)",0.15],["Morph (fp8)",0.137],["DeepInfra (fp4)",0.125],["SiliconFlow (fp8)",0.13],["InferenceNet",0.07],["Makora (fp4)",0.19],["AkashML (fp8)",0.19],["Phala",0.156],["Inceptron (fp4)",0.2],["DigitalOcean",0.169],["GMICloud (fp8)",0.182],["Alibaba",0.238],["Decart (fp4)",0.1955],["Wafer",0.14],["Friendli",0.234],["AtlasCloud (fp8)",0.26],["Baidu (fp8)",0.26],["BaseTen (fp4)",0.14],["Cloudflare",0.26],["Crusoe (fp4)",0.26],["Fireworks",0.26],["Mistral (nvfp4)",0.14],["Modal",0.26],["Parasail (fp8)",0.26],["PrimeIntellect",0.26],["Together",0.26],["Venice",0.26],["Z.AI (fp8)",0.26],["Relace",0.03]]}]]},"note":"Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.","factContext":"GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06","sourceIds":["openrouter-api-snapshot"]}}]