{"unit":"usd_per_1m_tokens","collected_at":"2026-09-15T06:37:12Z","sources":["openrouter","openrouter-endpoints"],"source":"OpenRouter API (live) + routes d'accès par modèle (219 modèles)","models":[{"id":"mistralai/mistral-nemo","slug":"mistral-nemo","name":"Mistral Nemo","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.019,"output":0.03,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":16384,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","url":"https://openrouter.ai/mistralai/mistral-nemo","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DekaLLM","tag":"dekallm/fp8","kind":"host","input":0.018,"output":0.03,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.019,"output":0.03,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.03,"output":0.03,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.04,"output":0.17,"cache_read":null,"context":60288,"quantization":"fp8"},{"provider":"Io Net","tag":"io-net/fp16","kind":"host","input":0.044,"output":0.16,"cache_read":0.029,"context":128000,"quantization":"fp16"}]},{"id":"ibm-granite/granite-4.0-h-micro","slug":"granite-4-0-h-micro","name":"Granite 4.0 Micro","vendor":"ibm-granite","vendor_name":"IBM","provider":"OpenRouter","provider_id":"openrouter","family":"granite","input":0.017,"output":0.112,"context":131000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":117900,"speed":null,"modalities":["text"],"features":["json"],"description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","url":"https://openrouter.ai/ibm-granite/granite-4.0-h-micro","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.017,"output":0.112,"cache_read":null,"context":131000,"quantization":null}]},{"id":"openai/gpt-oss-20b","slug":"gpt-oss-20b","name":"gpt-oss-20b","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.03,"output":0.13,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.03,"max_output":117964,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","url":"https://openrouter.ai/openai/gpt-oss-20b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Darkbloom","tag":"darkbloom/fp8","kind":"host","input":0.02,"output":0.1,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"AkashML","tag":"akashml/fp4","kind":"host","input":0.02,"output":0.1,"cache_read":null,"context":131072,"quantization":"fp4"},{"provider":"CoreWeave","tag":"coreweave/fp4","kind":"host","input":0.03,"output":0.13,"cache_read":0.03,"context":131072,"quantization":"fp4"},{"provider":"DekaLLM","tag":"dekallm/bf16","kind":"host","input":0.029,"output":0.14,"cache_read":null,"context":131072,"quantization":"bf16"},{"provider":"DeepInfra","tag":"deepinfra/bf16","kind":"host","input":0.03,"output":0.14,"cache_read":null,"context":131072,"quantization":"bf16"},{"provider":"Parasail","tag":"parasail/fp4","kind":"host","input":0.03,"output":0.15,"cache_read":0.02,"context":131072,"quantization":"fp4"},{"provider":"Phala","tag":"phala","kind":"host","input":0.04,"output":0.15,"cache_read":null,"context":131072,"quantization":null},{"provider":"Novita","tag":"novita/fp4","kind":"host","input":0.04,"output":0.15,"cache_read":null,"context":131072,"quantization":"fp4"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.04,"output":0.18,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"Together","tag":"together","kind":"host","input":0.05,"output":0.2,"cache_read":null,"context":131072,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/eu-west-1","kind":"cloud","input":0.07,"output":0.15,"cache_read":null,"context":131072,"quantization":null},{"provider":"Google","tag":"google-vertex/us-central1","kind":"cloud","input":0.07,"output":0.25,"cache_read":null,"context":131072,"quantization":null},{"provider":"Groq","tag":"groq","kind":"host","input":0.075,"output":0.3,"cache_read":0.0375,"context":131072,"quantization":null}]},{"id":"qwen/qwen3.7-flash","slug":"qwen3-7-flash","name":"Qwen3.7 Flash","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.03,"output":0.13,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.006,"max_output":65536,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","url":"https://openrouter.ai/qwen/qwen3.7-flash","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.03,"output":0.13,"cache_read":0.006,"context":1000000,"quantization":null}]},{"id":"meta-llama/llama-3.1-8b-instruct","slug":"llama-3-1-8b-instruct","name":"Llama 3.1 8B Instruct","vendor":"meta-llama","vendor_name":"Meta","provider":"OpenRouter","provider_id":"openrouter","family":"llama","input":0.05,"output":0.08,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.025,"max_output":117964,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","url":"https://openrouter.ai/meta-llama/llama-3.1-8b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.02,"output":0.04,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.02,"output":0.05,"cache_read":null,"context":16384,"quantization":"fp8"},{"provider":"Groq","tag":"groq","kind":"host","input":0.05,"output":0.08,"cache_read":0.025,"context":131072,"quantization":null},{"provider":"Cloudflare","tag":"cloudflare/fp8","kind":"host","input":0.152,"output":0.287,"cache_read":null,"context":32000,"quantization":"fp8"},{"provider":"CoreWeave","tag":"coreweave/bf16","kind":"host","input":0.22,"output":0.22,"cache_read":0.22,"context":131072,"quantization":"bf16"}]},{"id":"mistralai/mistral-small-24b-instruct-2501","slug":"mistral-small-24b-instruct-2501","name":"Mistral Small 3","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.05,"output":0.08,"context":32768,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":16384,"speed":null,"modalities":["text"],"features":["json"],"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","url":"https://openrouter.ai/mistralai/mistral-small-24b-instruct-2501","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.05,"output":0.08,"cache_read":null,"context":32768,"quantization":"fp8"}]},{"id":"amazon/nova-micro-v1","slug":"nova-micro-v1","name":"Nova Micro 1.0","vendor":"amazon","vendor_name":"Amazon","provider":"OpenRouter","provider_id":"openrouter","family":"nova","input":0.035,"output":0.14,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":5120,"speed":null,"modalities":["text"],"features":["tools"],"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","url":"https://openrouter.ai/amazon/nova-micro-v1","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Amazon Bedrock","tag":"amazon-bedrock/eu-west-1","kind":"cloud","input":0.035,"output":0.14,"cache_read":null,"context":128000,"quantization":null}]},{"id":"google/gemma-3-4b-it","slug":"gemma-3-4b-it","name":"Gemma 3 4B","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemma","input":0.05,"output":0.1,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":16384,"speed":null,"modalities":["text","image"],"features":["json"],"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","url":"https://openrouter.ai/google/gemma-3-4b-it","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/bf16","kind":"host","input":0.05,"output":0.1,"cache_read":null,"context":131072,"quantization":"bf16"}]},{"id":"cohere/command-r7b-12-2024","slug":"command-r7b-12-2024","name":"Command R7B (12-2024)","vendor":"cohere","vendor_name":"Cohere","provider":"OpenRouter","provider_id":"openrouter","family":"command","input":0.0375,"output":0.15,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":4000,"speed":null,"modalities":["text"],"features":["json"],"description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","url":"https://openrouter.ai/cohere/command-r7b-12-2024","direct_input":0.0375,"direct_output":0.15,"direct_cache_read":null,"direct_source":"cohere","routes":[{"provider":"Cohere","tag":"cohere","kind":"official","input":0.0375,"output":0.15,"cache_read":null,"context":128000,"quantization":null}]},{"id":"deepseek/deepseek-v4-flash-0731","slug":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.055,"output":0.11,"context":1310720,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.011,"max_output":943718,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","url":"https://openrouter.ai/deepseek/deepseek-v4-flash-0731","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"OpenInference","tag":"open-inference/fp8","kind":"host","input":0.04,"output":0.1,"cache_read":0.01,"context":1048576,"quantization":"fp8"},{"provider":"Relace","tag":"relace/fp4","kind":"host","input":0.055,"output":0.11,"cache_read":0.011,"context":1048576,"quantization":"fp4"},{"provider":"StreamLake","tag":"streamlake/fp8","kind":"host","input":0.0572,"output":0.1716,"cache_read":0.0018,"context":1024000,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.06,"output":0.18,"cache_read":0.015,"context":1048576,"quantization":"fp8"},{"provider":"Inceptron","tag":"inceptron/fp4","kind":"host","input":0.064,"output":0.1734,"cache_read":0.0102,"context":1048576,"quantization":"fp4"},{"provider":"Makora","tag":"makora","kind":"host","input":0.09,"output":0.195,"cache_read":0.0196,"context":1000000,"quantization":null},{"provider":"Wafer","tag":"wafer/fast","kind":"host","input":0.1,"output":0.25,"cache_read":0.05,"context":1048576,"quantization":null},{"provider":"Sail Research","tag":"sail-research/fp4","kind":"host","input":0.0741,"output":0.342,"cache_read":0.0228,"context":1048576,"quantization":"fp4"},{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":0.119,"output":0.238,"cache_read":0.0238,"context":1048576,"quantization":null},{"provider":"BaseTen","tag":"baseten/fp8","kind":"host","input":0.13,"output":0.26,"cache_read":0.028,"context":1048576,"quantization":"fp8"},{"provider":"CoreWeave","tag":"coreweave/fp8","kind":"host","input":0.13,"output":0.28,"cache_read":0.07,"context":262144,"quantization":"fp8"},{"provider":"Together","tag":"together","kind":"host","input":0.14,"output":0.28,"cache_read":0.03,"context":1048576,"quantization":null},{"provider":"Morph","tag":"morph/bf16","kind":"host","input":0.1234,"output":0.3475,"cache_read":0.0312,"context":1048576,"quantization":"bf16"},{"provider":"Venice","tag":"venice","kind":"host","input":0.175,"output":0.35,"cache_read":0.035,"context":1000000,"quantization":null},{"provider":"Reka","tag":"reka/fp4","kind":"host","input":0.11,"output":0.66,"cache_read":0.007,"context":262144,"quantization":"fp4"},{"provider":"Mancer 2","tag":"mancer/fp8","kind":"host","input":0.2,"output":0.6,"cache_read":null,"context":1048576,"quantization":"fp8"},{"provider":"Fireworks","tag":"fireworks","kind":"host","input":0.22,"output":0.66,"cache_read":0.007,"context":1048576,"quantization":null},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.22,"output":0.66,"cache_read":0.028,"context":1048576,"quantization":"fp8"},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.286,"output":0.858,"cache_read":0.0091,"context":1048575,"quantization":"fp8"},{"provider":"NextBit","tag":"nextbit/fp8","kind":"host","input":0.352,"output":1.056,"cache_read":0.012,"context":1048576,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.352,"output":1.056,"cache_read":0.0352,"context":1000000,"quantization":null},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.4092,"output":1.2276,"cache_read":0.026,"context":1048576,"quantization":"fp8"},{"provider":"Phala","tag":"phala","kind":"host","input":0.44,"output":1.32,"cache_read":0.028,"context":1048576,"quantization":null},{"provider":"Baidu","tag":"baidu/fp8","kind":"host","input":0.44,"output":1.32,"cache_read":0.014,"context":1048576,"quantization":"fp8"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp4","kind":"host","input":0.44,"output":1.32,"cache_read":0.028,"context":1048576,"quantization":"fp4"},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.44,"output":1.32,"cache_read":0.014,"context":1310720,"quantization":null}]},{"id":"openai/gpt-oss-120b","slug":"gpt-oss-120b","name":"gpt-oss-120b","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.037,"output":0.17,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":117964,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","url":"https://openrouter.ai/openai/gpt-oss-120b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"AkashML","tag":"akashml/bf16","kind":"host","input":0.03,"output":0.17,"cache_read":0.03,"context":131072,"quantization":"bf16"},{"provider":"CoreWeave","tag":"coreweave/fp4","kind":"host","input":0.03,"output":0.17,"cache_read":0.03,"context":131072,"quantization":"fp4"},{"provider":"DekaLLM","tag":"dekallm/bf16","kind":"host","input":0.03,"output":0.18,"cache_read":null,"context":131072,"quantization":"bf16"},{"provider":"DeepInfra","tag":"deepinfra/bf16","kind":"host","input":0.037,"output":0.17,"cache_read":null,"context":131072,"quantization":"bf16"},{"provider":"Crusoe","tag":"crusoe/bf16","kind":"host","input":0.05,"output":0.25,"cache_read":0.05,"context":131072,"quantization":"bf16"},{"provider":"Novita","tag":"novita/fp4","kind":"host","input":0.05,"output":0.25,"cache_read":null,"context":131072,"quantization":"fp4"},{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":0.06,"output":0.42,"cache_read":0.012,"context":128000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":0.09,"output":0.36,"cache_read":null,"context":131072,"quantization":null},{"provider":"Mancer 2","tag":"mancer/fp8","kind":"host","input":0.055,"output":0.5,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"BaseTen","tag":"baseten/fp4","kind":"host","input":0.1,"output":0.5,"cache_read":0.1,"context":128072,"quantization":"fp4"},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/eu-west-1","kind":"cloud","input":0.15,"output":0.6,"cache_read":null,"context":131072,"quantization":null},{"provider":"Nebius","tag":"nebius/fp4","kind":"host","input":0.15,"output":0.6,"cache_read":null,"context":131072,"quantization":"fp4"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.15,"output":0.6,"cache_read":0.075,"context":131072,"quantization":"fp8"},{"provider":"Phala","tag":"phala","kind":"host","input":0.15,"output":0.6,"cache_read":null,"context":131072,"quantization":null},{"provider":"Together","tag":"together","kind":"host","input":0.15,"output":0.6,"cache_read":null,"context":131072,"quantization":null},{"provider":"Groq","tag":"groq","kind":"host","input":0.15,"output":0.6,"cache_read":0.075,"context":131072,"quantization":null},{"provider":"Parasail","tag":"parasail/fp4","kind":"host","input":0.1,"output":0.75,"cache_read":0.055,"context":131072,"quantization":"fp4"},{"provider":"Mara","tag":"mara","kind":"host","input":0.15,"output":0.75,"cache_read":null,"context":131072,"quantization":null},{"provider":"SambaNova","tag":"sambanova","kind":"host","input":0.14,"output":0.95,"cache_read":null,"context":131072,"quantization":null},{"provider":"Cerebras","tag":"cerebras/fp16","kind":"host","input":0.35,"output":0.75,"cache_read":0.35,"context":131072,"quantization":"fp16"}]},{"id":"meta-llama/llama-3.2-1b-instruct","slug":"llama-3-2-1b-instruct","name":"Llama 3.2 1B Instruct","vendor":"meta-llama","vendor_name":"Meta","provider":"OpenRouter","provider_id":"openrouter","family":"llama","input":0.027,"output":0.201,"context":60000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":54000,"speed":null,"modalities":["text"],"features":[],"description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","url":"https://openrouter.ai/meta-llama/llama-3.2-1b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.027,"output":0.201,"cache_read":null,"context":60000,"quantization":null}]},{"id":"google/gemma-3-12b-it","slug":"gemma-3-12b-it","name":"Gemma 3 12B","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemma","input":0.05,"output":0.15,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":16384,"speed":null,"modalities":["text","image"],"features":["json","tools"],"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","url":"https://openrouter.ai/google/gemma-3-12b-it","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/bf16","kind":"host","input":0.05,"output":0.15,"cache_read":null,"context":131072,"quantization":"bf16"}]},{"id":"qwen/qwen3-30b-a3b-instruct-2507","slug":"qwen3-30b-a3b-instruct-2507","name":"Qwen3 30B A3B Instruct 2507","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.0481,"output":0.193,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":32000,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","url":"https://openrouter.ai/qwen/qwen3-30b-a3b-instruct-2507","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"StreamLake","tag":"streamlake","kind":"host","input":0.0481,"output":0.193,"cache_read":null,"context":128000,"quantization":null},{"provider":"DekaLLM","tag":"dekallm","kind":"host","input":0.09,"output":0.3,"cache_read":null,"context":262144,"quantization":null},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.09,"output":0.3,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Nebius","tag":"nebius/fp8","kind":"host","input":0.1,"output":0.3,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.13,"output":0.52,"cache_read":null,"context":131072,"quantization":null}]},{"id":"microsoft/phi-4","slug":"phi-4","name":"Phi 4","vendor":"microsoft","vendor_name":"Microsoft","provider":"OpenRouter","provider_id":"openrouter","family":"phi","input":0.07,"output":0.14,"context":16384,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":14745,"speed":null,"modalities":["text"],"features":["json"],"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","url":"https://openrouter.ai/microsoft/phi-4","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/bf16","kind":"host","input":0.07,"output":0.14,"cache_read":null,"context":16384,"quantization":"bf16"}]},{"id":"mistralai/ministral-3b-2512","slug":"ministral-3b-2512","name":"Ministral 3 3B 2512","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.1,"output":0.1,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.01,"max_output":104857,"speed":null,"modalities":["text","image"],"features":["json","tools"],"description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","url":"https://openrouter.ai/mistralai/ministral-3b-2512","direct_input":0.1,"direct_output":0.1,"direct_cache_read":0.01,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/zdr","kind":"official","input":0.1,"output":0.1,"cache_read":0.01,"context":131072,"quantization":null}]},{"id":"amazon/nova-lite-v1","slug":"nova-lite-v1","name":"Nova Lite 1.0","vendor":"amazon","vendor_name":"Amazon","provider":"OpenRouter","provider_id":"openrouter","family":"nova","input":0.06,"output":0.24,"context":300000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":5120,"speed":null,"modalities":["text","image"],"features":["tools"],"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","url":"https://openrouter.ai/amazon/nova-lite-v1","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Amazon Bedrock","tag":"amazon-bedrock/eu-west-1","kind":"cloud","input":0.06,"output":0.24,"cache_read":null,"context":300000,"quantization":null}]},{"id":"mistralai/mistral-small-3.2-24b-instruct","slug":"mistral-small-3-2-24b-instruct","name":"Mistral Small 3.2 24B","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.075,"output":0.2,"context":256000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":16384,"speed":null,"modalities":["image","text"],"features":["json","tools"],"description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","url":"https://openrouter.ai/mistralai/mistral-small-3.2-24b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.075,"output":0.2,"cache_read":null,"context":128000,"quantization":"fp8"},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.0938,"output":0.25,"cache_read":null,"context":256000,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/bf16","kind":"host","input":0.09,"output":0.3,"cache_read":0.05,"context":131072,"quantization":"bf16"}]},{"id":"ibm-granite/granite-4.2-8b","slug":"granite-4-2-8b","name":"Granite 4.2 8B","vendor":"ibm-granite","vendor_name":"IBM","provider":"OpenRouter","provider_id":"openrouter","family":"granite","input":0.06,"output":0.25,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.015,"max_output":117964,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Granite 4.2 8B is a dense reasoning model from IBM. It is suited for mathematics, code generation, multilingual dialogue, and agentic workflows that need multi-step reasoning. It supports full, low-effort,...","url":"https://openrouter.ai/ibm-granite/granite-4.2-8b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/bf16","kind":"host","input":0.06,"output":0.25,"cache_read":0.015,"context":131072,"quantization":"bf16"},{"provider":"CoreWeave","tag":"coreweave/bf16","kind":"host","input":0.1,"output":0.15,"cache_read":0.05,"context":131072,"quantization":"bf16"}]},{"id":"deepseek/deepseek-v4-flash","slug":"deepseek-v4-flash","name":"DeepSeek V4 Flash 0423","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.0886,"output":0.1772,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.0177,"max_output":384000,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","url":"https://openrouter.ai/deepseek/deepseek-v4-flash","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"OpenInference","tag":"open-inference/fp8","kind":"host","input":0.05,"output":0.14,"cache_read":0.014,"context":1048576,"quantization":"fp8"},{"provider":"StreamLake","tag":"streamlake/fp8","kind":"host","input":0.0886,"output":0.1772,"cache_read":0.0177,"context":1024000,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.09,"output":0.18,"cache_read":0.018,"context":1048576,"quantization":"fp8"},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.091,"output":0.182,"cache_read":0.0182,"context":1048575,"quantization":"fp8"},{"provider":"Venice","tag":"venice","kind":"host","input":0.0966,"output":0.1925,"cache_read":0.0196,"context":1000000,"quantization":null},{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":0.098,"output":0.196,"cache_read":0.0196,"context":1048576,"quantization":null},{"provider":"Wafer","tag":"wafer","kind":"host","input":0.1,"output":0.25,"cache_read":0.05,"context":1048576,"quantization":null},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.13,"output":0.28,"cache_read":0.028,"context":1048576,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba/fp8","kind":"host","input":0.134,"output":0.268,"cache_read":0.0268,"context":1000000,"quantization":"fp8"},{"provider":"Baidu","tag":"baidu/fp8","kind":"host","input":0.14,"output":0.28,"cache_read":0.028,"context":1048576,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.14,"output":0.28,"cache_read":0.028,"context":1048576,"quantization":"fp8"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp4","kind":"host","input":0.14,"output":0.28,"cache_read":0.028,"context":1048576,"quantization":"fp4"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.14,"output":0.28,"cache_read":0.07,"context":1048576,"quantization":"fp8"},{"provider":"NextBit","tag":"nextbit/fp8","kind":"host","input":0.15,"output":0.35,"cache_read":0.035,"context":1048576,"quantization":"fp8"},{"provider":"Phala","tag":"phala","kind":"host","input":0.2,"output":0.4,"cache_read":0.07,"context":1048576,"quantization":null},{"provider":"Mancer 2","tag":"mancer/fp8","kind":"host","input":0.19,"output":0.5,"cache_read":null,"context":1048576,"quantization":"fp8"},{"provider":"Azure","tag":"azure/us","kind":"cloud","input":0.21,"output":0.56,"cache_read":0.031,"context":1048576,"quantization":null}]},{"id":"qwen/qwen3.5-9b","slug":"qwen3-5-9b","name":"Qwen3.5-9B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.1,"output":0.15,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":235929,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","url":"https://openrouter.ai/qwen/qwen3.5-9b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Darkbloom","tag":"darkbloom/fp4","kind":"host","input":0.08,"output":0.13,"cache_read":null,"context":262144,"quantization":"fp4"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.1,"output":0.15,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/bf16","kind":"host","input":0.1,"output":0.15,"cache_read":null,"context":262144,"quantization":"bf16"},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.1,"output":0.15,"cache_read":null,"context":256000,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/bf16","kind":"host","input":0.1,"output":0.25,"cache_read":null,"context":262144,"quantization":"bf16"},{"provider":"Together","tag":"together","kind":"host","input":0.17,"output":0.25,"cache_read":null,"context":262144,"quantization":null}]},{"id":"qwen/qwen3.5-flash-02-23","slug":"qwen3-5-flash-02-23","name":"Qwen3.5-Flash","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.065,"output":0.26,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":65536,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","url":"https://openrouter.ai/qwen/qwen3.5-flash-02-23","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.065,"output":0.26,"cache_read":null,"context":1000000,"quantization":null}]},{"id":"meta-llama/llama-3.2-3b-instruct","slug":"llama-3-2-3b-instruct","name":"Llama 3.2 3B Instruct","vendor":"meta-llama","vendor_name":"Meta","provider":"OpenRouter","provider_id":"openrouter","family":"llama","input":0.05,"output":0.33,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":117964,"speed":null,"modalities":["text"],"features":["json"],"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","url":"https://openrouter.ai/meta-llama/llama-3.2-3b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Parasail","tag":"parasail/bf16","kind":"host","input":0.05,"output":0.33,"cache_read":null,"context":131072,"quantization":"bf16"},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.0509,"output":0.335,"cache_read":null,"context":80000,"quantization":null}]},{"id":"qwen/qwen3-coder-30b-a3b-instruct","slug":"qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30B A3B Instruct","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.07,"output":0.28,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":235929,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","url":"https://openrouter.ai/qwen/qwen3-coder-30b-a3b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.07,"output":0.27,"cache_read":null,"context":160000,"quantization":"fp8"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.07,"output":0.28,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":0.15,"output":0.6,"cache_read":null,"context":0,"quantization":null},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.2925,"output":1.4625,"cache_read":null,"context":262144,"quantization":null}]},{"id":"qwen/qwen-2.5-7b-instruct","slug":"qwen-2-5-7b-instruct","name":"Qwen2.5 7B Instruct","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.1,"output":0.2,"context":32768,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":29491,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","url":"https://openrouter.ai/qwen/qwen-2.5-7b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Phala","tag":"phala","kind":"host","input":0.1,"output":0.2,"cache_read":null,"context":32768,"quantization":null}]},{"id":"qwen/qwen3-32b","slug":"qwen3-32b","name":"Qwen3 32B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.08,"output":0.28,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":16384,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","url":"https://openrouter.ai/qwen/qwen3-32b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.08,"output":0.28,"cache_read":null,"context":40960,"quantization":"fp8"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.14,"output":0.57,"cache_read":null,"context":131072,"quantization":"fp8"}]},{"id":"openai/gpt-oss-safeguard-20b","slug":"gpt-oss-safeguard-20b","name":"gpt-oss-safeguard-20b","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.075,"output":0.3,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.0375,"max_output":65536,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","url":"https://openrouter.ai/openai/gpt-oss-safeguard-20b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Groq","tag":"groq","kind":"host","input":0.075,"output":0.3,"cache_read":0.0375,"context":131072,"quantization":null}]},{"id":"openai/gpt-5-nano","slug":"gpt-5-nano","name":"GPT-5 Nano","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.05,"output":0.4,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.005,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","url":"https://openrouter.ai/openai/gpt-5-nano","direct_input":0.025,"direct_output":0.2,"direct_cache_read":0.0025,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":0.025,"output":0.2,"cache_read":0.0025,"context":400000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":0.05,"output":0.4,"cache_read":0.01,"context":400000,"quantization":null}]},{"id":"google/gemma-4-26b-a4b-it","slug":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemma","input":0.09,"output":0.3,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.05,"max_output":235929,"speed":null,"modalities":["image","text","video"],"features":["json","reasoning","tools"],"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","url":"https://openrouter.ai/google/gemma-4-26b-a4b-it","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Darkbloom","tag":"darkbloom","kind":"host","input":0.042,"output":0.22,"cache_read":null,"context":131072,"quantization":null},{"provider":"DekaLLM","tag":"dekallm/bf16","kind":"host","input":0.06,"output":0.33,"cache_read":null,"context":262144,"quantization":"bf16"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.07,"output":0.34,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"NextBit","tag":"nextbit/bf16","kind":"host","input":0.09,"output":0.3,"cache_read":0.05,"context":262144,"quantization":"bf16"},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.1,"output":0.3,"cache_read":null,"context":256000,"quantization":null},{"provider":"Makora","tag":"makora","kind":"host","input":0.1,"output":0.34,"cache_read":0.034,"context":262144,"quantization":null},{"provider":"Venice","tag":"venice/bf16","kind":"host","input":0.13,"output":0.4,"cache_read":0.05,"context":256000,"quantization":"bf16"},{"provider":"Parasail","tag":"parasail/bf16","kind":"host","input":0.13,"output":0.4,"cache_read":0.05,"context":262144,"quantization":"bf16"},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.13,"output":0.4,"cache_read":null,"context":262144,"quantization":"bf16"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.14,"output":0.4,"cache_read":0.05,"context":262144,"quantization":"fp8"},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":0.15,"output":0.6,"cache_read":null,"context":262144,"quantization":null}]},{"id":"z-ai/glm-4.7-flash","slug":"glm-4-7-flash","name":"GLM 4.7 Flash","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":0.0605,"output":0.4,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":117964,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","url":"https://openrouter.ai/z-ai/glm-4.7-flash","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.06,"output":0.4,"cache_read":0.01,"context":128000,"quantization":"fp8"},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.0605,"output":0.4,"cache_read":null,"context":131072,"quantization":null},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.07,"output":0.4,"cache_read":0.01,"context":200000,"quantization":"bf16"}]},{"id":"mistralai/ministral-8b-2512","slug":"ministral-8b-2512","name":"Ministral 3 8B 2512","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.15,"output":0.15,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.015,"max_output":209715,"speed":null,"modalities":["text","image"],"features":["json","tools"],"description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","url":"https://openrouter.ai/mistralai/ministral-8b-2512","direct_input":0.15,"direct_output":0.15,"direct_cache_read":0.015,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/zdr","kind":"official","input":0.15,"output":0.15,"cache_read":0.015,"context":262144,"quantization":null}]},{"id":"qwen/qwen3-14b","slug":"qwen3-14b","name":"Qwen3 14B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.12,"output":0.24,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":16384,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","url":"https://openrouter.ai/qwen/qwen3-14b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"NextBit","tag":"nextbit/int4","kind":"host","input":0.1,"output":0.22,"cache_read":null,"context":40960,"quantization":"int4"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.12,"output":0.24,"cache_read":null,"context":40960,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.2275,"output":0.91,"cache_read":null,"context":131072,"quantization":null}]},{"id":"meta-llama/llama-4-scout","slug":"llama-4-scout","name":"Llama 4 Scout","vendor":"meta-llama","vendor_name":"Meta","provider":"OpenRouter","provider_id":"openrouter","family":"llama","input":0.1,"output":0.3,"context":1310720,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":16384,"speed":null,"modalities":["text","image"],"features":["json","tools"],"description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","url":"https://openrouter.ai/meta-llama/llama-4-scout","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.1,"output":0.3,"cache_read":null,"context":327680,"quantization":"fp8"},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.18,"output":0.59,"cache_read":null,"context":131072,"quantization":"bf16"},{"provider":"Google","tag":"google-vertex/us-east5","kind":"cloud","input":0.25,"output":0.7,"cache_read":null,"context":1310720,"quantization":null}]},{"id":"google/gemma-4-31b-it","slug":"gemma-4-31b-it","name":"Gemma 4 31B","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemma","input":0.09,"output":0.34,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.05,"max_output":16384,"speed":null,"modalities":["image","text","video"],"features":["json","reasoning","tools"],"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","url":"https://openrouter.ai/google/gemma-4-31b-it","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/turbo","kind":"host","input":0.09,"output":0.34,"cache_read":0.05,"context":262144,"quantization":"fp4"},{"provider":"CoreWeave","tag":"coreweave/fp4","kind":"host","input":0.1,"output":0.34,"cache_read":0.1,"context":262144,"quantization":"fp4"},{"provider":"Venice","tag":"venice/bf16","kind":"host","input":0.12,"output":0.36,"cache_read":0.09,"context":256000,"quantization":"bf16"},{"provider":"Chutes","tag":"chutes/fp4","kind":"host","input":0.12,"output":0.37,"cache_read":0.012,"context":131072,"quantization":"fp4"},{"provider":"Crusoe","tag":"crusoe","kind":"host","input":0.14,"output":0.4,"cache_read":0.14,"context":262144,"quantization":null},{"provider":"Friendli","tag":"friendli","kind":"host","input":0.14,"output":0.4,"cache_read":null,"context":262144,"quantization":null},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.14,"output":0.4,"cache_read":null,"context":262144,"quantization":"bf16"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.15,"output":0.4,"cache_read":0.06,"context":262144,"quantization":"fp8"},{"provider":"Together","tag":"together","kind":"host","input":0.39,"output":0.97,"cache_read":null,"context":262144,"quantization":null},{"provider":"SambaNova","tag":"sambanova","kind":"host","input":0.38,"output":1.15,"cache_read":null,"context":131072,"quantization":null},{"provider":"ModelRun","tag":"modelrun/fp4","kind":"host","input":0.75,"output":1,"cache_read":0.75,"context":262144,"quantization":"fp4"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.75,"output":1,"cache_read":0.25,"context":262144,"quantization":"fp8"}]},{"id":"qwen/qwen3-235b-a22b-2507","slug":"qwen3-235b-a22b-2507","name":"Qwen3 235B A22B Instruct 2507","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.0875,"output":0.35,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.0175,"max_output":235929,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","url":"https://openrouter.ai/qwen/qwen3-235b-a22b-2507","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.0875,"output":0.35,"cache_read":0.0175,"context":262144,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.09,"output":0.55,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.09,"output":0.58,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.1495,"output":0.598,"cache_read":null,"context":131072,"quantization":null},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.15,"output":0.75,"cache_read":null,"context":128000,"quantization":"fp8"},{"provider":"Nebius","tag":"nebius/fp8","kind":"host","input":0.2,"output":0.6,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.14,"output":0.8,"cache_read":0.05,"context":131072,"quantization":"fp8"},{"provider":"StreamLake","tag":"streamlake","kind":"host","input":0.21,"output":0.84,"cache_read":null,"context":128000,"quantization":null},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.2,"output":0.88,"cache_read":0.2,"context":131072,"quantization":"fp8"},{"provider":"Google","tag":"google-vertex/us-south1","kind":"cloud","input":0.22,"output":0.88,"cache_read":null,"context":262144,"quantization":null}]},{"id":"meta-llama/llama-3.3-70b-instruct","slug":"llama-3-3-70b-instruct","name":"Llama 3.3 70B Instruct","vendor":"meta-llama","vendor_name":"Meta","provider":"OpenRouter","provider_id":"openrouter","family":"llama","input":0.1,"output":0.32,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":16384,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","url":"https://openrouter.ai/meta-llama/llama-3.3-70b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/turbo","kind":"host","input":0.1,"output":0.32,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.135,"output":0.4,"cache_read":null,"context":12288,"quantization":"bf16"},{"provider":"AkashML","tag":"akashml/fp8","kind":"host","input":0.2,"output":0.52,"cache_read":0.1,"context":131072,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.22,"output":0.5,"cache_read":0.11,"context":131072,"quantization":"fp8"},{"provider":"SambaNova","tag":"sambanova-turbo","kind":"host","input":0.45,"output":0.9,"cache_read":null,"context":131072,"quantization":null},{"provider":"Groq","tag":"groq","kind":"host","input":0.59,"output":0.79,"cache_read":0.295,"context":131072,"quantization":null},{"provider":"CoreWeave","tag":"coreweave/fp16","kind":"host","input":0.71,"output":0.71,"cache_read":0.71,"context":128000,"quantization":"fp16"},{"provider":"Google","tag":"google-vertex/us-central1","kind":"cloud","input":0.72,"output":0.72,"cache_read":null,"context":128000,"quantization":null},{"provider":"Cloudflare","tag":"cloudflare/fp8","kind":"host","input":0.293,"output":2.253,"cache_read":null,"context":24000,"quantization":"fp8"},{"provider":"Together","tag":"together","kind":"host","input":1.04,"output":1.04,"cache_read":null,"context":131072,"quantization":null}]},{"id":"google/gemma-3-27b-it","slug":"gemma-3-27b-it","name":"Gemma 3 27B","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemma","input":0.08,"output":0.45,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.04,"max_output":117964,"speed":null,"modalities":["text","image"],"features":["json","tools"],"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","url":"https://openrouter.ai/google/gemma-3-27b-it","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.08,"output":0.16,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.119,"output":0.2,"cache_read":null,"context":98304,"quantization":"bf16"},{"provider":"Nebius","tag":"nebius/fp8","kind":"host","input":0.1,"output":0.3,"cache_read":null,"context":110000,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.08,"output":0.45,"cache_read":0.04,"context":131072,"quantization":"fp8"}]},{"id":"google/gemini-2.5-flash-lite","slug":"gemini-2-5-flash-lite","name":"Gemini 2.5 Flash Lite","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":0.1,"output":0.4,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.01,"max_output":65535,"speed":null,"modalities":["text","image","file","audio","video"],"features":["json","reasoning","tools"],"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","url":"https://openrouter.ai/google/gemini-2.5-flash-lite","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.05,"output":0.2,"cache_read":0.005,"context":1048576,"quantization":null},{"provider":"Google","tag":"google-vertex/eu","kind":"cloud","input":0.1,"output":0.4,"cache_read":0.01,"context":1048576,"quantization":null}]},{"id":"openai/gpt-4.1-nano","slug":"gpt-4-1-nano","name":"GPT-4.1 Nano","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.1,"output":0.4,"context":1047576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.025,"max_output":32768,"speed":null,"modalities":["image","text","file"],"features":["json","tools"],"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","url":"https://openrouter.ai/openai/gpt-4.1-nano","direct_input":0.1,"direct_output":0.4,"direct_cache_read":0.025,"direct_source":"openai","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":0.1,"output":0.4,"cache_read":0.03,"context":1047576,"quantization":null},{"provider":"OpenAI","tag":"openai","kind":"official","input":0.1,"output":0.4,"cache_read":0.025,"context":1047576,"quantization":null}]},{"id":"qwen/qwen3-vl-32b-instruct","slug":"qwen3-vl-32b-instruct","name":"Qwen3 VL 32B Instruct","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.104,"output":0.416,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":32768,"speed":null,"modalities":["text","image"],"features":["json","tools"],"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","url":"https://openrouter.ai/qwen/qwen3-vl-32b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.104,"output":0.416,"cache_read":null,"context":131072,"quantization":null}]},{"id":"mistralai/ministral-14b-2512","slug":"ministral-14b-2512","name":"Ministral 3 14B 2512","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.2,"output":0.2,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.02,"max_output":209715,"speed":null,"modalities":["text","image"],"features":["json","tools"],"description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","url":"https://openrouter.ai/mistralai/ministral-14b-2512","direct_input":0.2,"direct_output":0.2,"direct_cache_read":0.02,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/zdr","kind":"official","input":0.2,"output":0.2,"cache_read":0.02,"context":262144,"quantization":null}]},{"id":"qwen/qwen3-8b","slug":"qwen3-8b","name":"Qwen3 8B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.117,"output":0.455,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":8192,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","url":"https://openrouter.ai/qwen/qwen3-8b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.117,"output":0.455,"cache_read":null,"context":131072,"quantization":null}]},{"id":"qwen/qwen3-vl-8b-instruct","slug":"qwen3-vl-8b-instruct","name":"Qwen3 VL 8B Instruct","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.117,"output":0.455,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":32768,"speed":null,"modalities":["image","text"],"features":["json","tools"],"description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","url":"https://openrouter.ai/qwen/qwen3-vl-8b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.117,"output":0.455,"cache_read":null,"context":131072,"quantization":null},{"provider":"Parasail","tag":"parasail/bf16","kind":"host","input":0.25,"output":0.75,"cache_read":0.12,"context":262144,"quantization":"bf16"}]},{"id":"qwen/qwen3-30b-a3b","slug":"qwen3-30b-a3b","name":"Qwen3 30B A3B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.12,"output":0.5,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":16384,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","url":"https://openrouter.ai/qwen/qwen3-30b-a3b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.12,"output":0.5,"cache_read":null,"context":40960,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.13,"output":0.52,"cache_read":null,"context":131072,"quantization":null}]},{"id":"qwen/qwen3.8-flash","slug":"qwen3-8-flash","name":"Qwen3.8 Flash","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.15,"output":0.47,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.016,"max_output":131072,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"Qwen3.8 Flash is a multimodal reasoning model from Alibaba. It is suited for coding assistance, agentic workflows, visual understanding, document and codebase analysis, desktop interaction, chart analysis, and long-video analysis.","url":"https://openrouter.ai/qwen/qwen3.8-flash","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Makora","tag":"makora/fp4","kind":"host","input":0.15,"output":0.47,"cache_read":0.016,"context":262144,"quantization":"fp4"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.15,"output":0.47,"cache_read":0.016,"context":1000000,"quantization":null}]},{"id":"z-ai/glm-5.3-flash","slug":"glm-5-3-flash","name":"GLM 5.3 Flash","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":0.15,"output":0.5,"context":1310720,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.03,"max_output":131072,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","url":"https://openrouter.ai/z-ai/glm-5.3-flash","direct_input":0.15,"direct_output":0.5,"direct_cache_read":0.03,"direct_source":"z-ai","routes":[{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.075,"output":0.25,"cache_read":0.015,"context":1048576,"quantization":"fp4"},{"provider":"Relace","tag":"relace","kind":"host","input":0.09,"output":0.3,"cache_read":0.018,"context":1048576,"quantization":null},{"provider":"Morph","tag":"morph/fp8","kind":"host","input":0.1,"output":0.35,"cache_read":0.02,"context":1048576,"quantization":"fp8"},{"provider":"Wafer","tag":"wafer","kind":"host","input":0.1,"output":0.35,"cache_read":0.02,"context":1048576,"quantization":null},{"provider":"StreamLake","tag":"streamlake/fp8","kind":"host","input":0.1123,"output":0.3745,"cache_read":0.0225,"context":1024000,"quantization":"fp8"},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.1125,"output":0.375,"cache_read":0.0225,"context":1048576,"quantization":"fp8"},{"provider":"Reka","tag":"reka/fp8","kind":"host","input":0.132,"output":0.44,"cache_read":0.0264,"context":262144,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.132,"output":0.44,"cache_read":0.0264,"context":1048576,"quantization":"fp8"},{"provider":"Makora","tag":"makora","kind":"host","input":0.14,"output":0.47,"cache_read":0.024,"context":1048576,"quantization":null},{"provider":"Crusoe","tag":"crusoe/fp4","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":"fp4"},{"provider":"CoreWeave","tag":"coreweave/fp8","kind":"host","input":0.15,"output":0.5,"cache_read":0.05,"context":1048576,"quantization":"fp8"},{"provider":"Sail Research","tag":"sail-research/fp8","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":"fp8"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":"fp8"},{"provider":"Fireworks","tag":"fireworks","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":null},{"provider":"Phala","tag":"phala/fp8","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":"fp8"},{"provider":"Friendli","tag":"friendli","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":null},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":"fp8"},{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":null},{"provider":"Together","tag":"together","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048575,"quantization":null},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":"fp8"},{"provider":"BaseTen","tag":"baseten/fp8","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":"fp8"},{"provider":"Venice","tag":"venice","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":null},{"provider":"Io Net","tag":"io-net/fp8","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":262144,"quantization":"fp8"},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.15,"output":0.5,"cache_read":0.03,"context":1310720,"quantization":null},{"provider":"Z.AI","tag":"z-ai/fp8","kind":"official","input":0.15,"output":0.5,"cache_read":0.03,"context":1048576,"quantization":"fp8"},{"provider":"NextBit","tag":"nextbit/fp8","kind":"host","input":0.177,"output":0.59,"cache_read":0.036,"context":1048576,"quantization":"fp8"},{"provider":"Modal","tag":"modal/fp8","kind":"host","input":0.45,"output":1.5,"cache_read":0.09,"context":1048576,"quantization":"fp8"}]},{"id":"cohere/command-r-08-2024","slug":"command-r-08-2024","name":"Command R (08-2024)","vendor":"cohere","vendor_name":"Cohere","provider":"OpenRouter","provider_id":"openrouter","family":"command","input":0.15,"output":0.6,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":4000,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","url":"https://openrouter.ai/cohere/command-r-08-2024","direct_input":0.15,"direct_output":0.6,"direct_cache_read":null,"direct_source":"cohere","routes":[{"provider":"Cohere","tag":"cohere","kind":"official","input":0.15,"output":0.6,"cache_read":null,"context":128000,"quantization":null}]},{"id":"mistralai/mistral-small-2603","slug":"mistral-small-2603","name":"Mistral Small 4","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.15,"output":0.6,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.015,"max_output":209715,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","url":"https://openrouter.ai/mistralai/mistral-small-2603","direct_input":0.15,"direct_output":0.6,"direct_cache_read":0.015,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/zdr","kind":"official","input":0.15,"output":0.6,"cache_read":0.015,"context":262144,"quantization":null},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.1875,"output":0.75,"cache_read":null,"context":256000,"quantization":"fp8"}]},{"id":"openai/gpt-4o-mini","slug":"gpt-4o-mini","name":"GPT-4o-mini","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.15,"output":0.6,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.075,"max_output":16384,"speed":null,"modalities":["text","image","file"],"features":["json","tools"],"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","url":"https://openrouter.ai/openai/gpt-4o-mini","direct_input":0.15,"direct_output":0.6,"direct_cache_read":0.075,"direct_source":"openai","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":0.15,"output":0.6,"cache_read":0.075,"context":128000,"quantization":null},{"provider":"OpenAI","tag":"openai","kind":"official","input":0.15,"output":0.6,"cache_read":0.075,"context":128000,"quantization":null}]},{"id":"openai/gpt-4o-mini-2024-07-18","slug":"gpt-4o-mini-2024-07-18","name":"GPT-4o-mini (2024-07-18)","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.15,"output":0.6,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.075,"max_output":16384,"speed":null,"modalities":["text","image","file"],"features":["json","tools"],"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","url":"https://openrouter.ai/openai/gpt-4o-mini-2024-07-18","direct_input":0.15,"direct_output":0.6,"direct_cache_read":0.075,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":0.15,"output":0.6,"cache_read":0.075,"context":128000,"quantization":null}]},{"id":"qwen/qwen3-vl-30b-a3b-instruct","slug":"qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.15,"output":0.6,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":16384,"speed":null,"modalities":["text","image"],"features":["json","tools"],"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","url":"https://openrouter.ai/qwen/qwen3-vl-30b-a3b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.13,"output":0.52,"cache_read":null,"context":131072,"quantization":null},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.15,"output":0.6,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.2,"output":0.7,"cache_read":null,"context":131072,"quantization":"bf16"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.29,"output":1,"cache_read":null,"context":262144,"quantization":"fp8"}]},{"id":"qwen/qwen3-coder-next","slug":"qwen3-coder-next","name":"Qwen3 Coder Next","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.12,"output":0.8,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.07,"max_output":235929,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","url":"https://openrouter.ai/qwen/qwen3-coder-next","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Parasail","tag":"parasail/bf16","kind":"host","input":0.12,"output":0.8,"cache_read":0.07,"context":262144,"quantization":"bf16"},{"provider":"StreamLake","tag":"streamlake","kind":"host","input":0.18,"output":0.9,"cache_read":0.036,"context":256000,"quantization":null},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.2,"output":1.5,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.3,"output":1.5,"cache_read":null,"context":262144,"quantization":null}]},{"id":"mistralai/mistral-saba","slug":"mistral-saba","name":"Saba","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.2,"output":0.6,"context":32768,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.02,"max_output":26214,"speed":null,"modalities":["text","file"],"features":["json","tools"],"description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","url":"https://openrouter.ai/mistralai/mistral-saba","direct_input":0.2,"direct_output":0.6,"direct_cache_read":0.02,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/zdr","kind":"official","input":0.2,"output":0.6,"cache_read":0.02,"context":32768,"quantization":null}]},{"id":"qwen/qwen3.6-35b-a3b","slug":"qwen3-6-35b-a3b","name":"Qwen3.6 35B A3B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.1,"output":0.9,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.05,"max_output":235929,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","url":"https://openrouter.ai/qwen/qwen3.6-35b-a3b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Darkbloom","tag":"darkbloom/fp4","kind":"host","input":0.05,"output":0.7,"cache_read":null,"context":262144,"quantization":"fp4"},{"provider":"AkashML","tag":"akashml/fp8","kind":"host","input":0.1,"output":0.9,"cache_read":0.05,"context":262144,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.1,"output":0.95,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"DekaLLM","tag":"dekallm","kind":"host","input":0.1,"output":1,"cache_read":0.05,"context":262144,"quantization":null},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.1,"output":1,"cache_read":null,"context":256000,"quantization":"fp8"},{"provider":"Io Net","tag":"io-net/fp8","kind":"host","input":0.133,"output":0.9405,"cache_read":0.0665,"context":262144,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.15,"output":1,"cache_read":0.05,"context":262144,"quantization":"fp8"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.186,"output":1.1138,"cache_read":0.186,"context":262144,"quantization":"fp8"},{"provider":"Phala","tag":"phala","kind":"host","input":0.2,"output":1.27,"cache_read":null,"context":262144,"quantization":null},{"provider":"CoreWeave","tag":"coreweave/fp8","kind":"host","input":0.25,"output":1.25,"cache_read":0.25,"context":262144,"quantization":"fp8"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.24,"output":1.8,"cache_read":0.15,"context":262144,"quantization":"fp8"}]},{"id":"deepseek/deepseek-v3.2","slug":"deepseek-v3-2","name":"DeepSeek V3.2","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.269,"output":0.4,"context":163840,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.1345,"max_output":65536,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","url":"https://openrouter.ai/deepseek/deepseek-v3.2","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.2088,"output":0.3096,"cache_read":0.0216,"context":163840,"quantization":"fp8"},{"provider":"StreamLake","tag":"streamlake/fp8","kind":"host","input":0.2145,"output":0.3217,"cache_read":0.0215,"context":128000,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.26,"output":0.38,"cache_read":0.13,"context":163840,"quantization":"fp4"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.26,"output":0.38,"cache_read":0.13,"context":163840,"quantization":"fp8"},{"provider":"Venice","tag":"venice","kind":"host","input":0.2683,"output":0.3902,"cache_read":0.1301,"context":160000,"quantization":null},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.259,"output":0.42,"cache_read":0.135,"context":163840,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.269,"output":0.4,"cache_read":0.1345,"context":163840,"quantization":"fp8"},{"provider":"Baidu","tag":"baidu/fp8","kind":"host","input":0.28,"output":0.42,"cache_read":0.028,"context":131072,"quantization":"fp8"},{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":0.3,"output":0.96,"cache_read":0.09,"context":163840,"quantization":null},{"provider":"Alibaba","tag":"alibaba/fp8","kind":"host","input":0.3705,"output":1.1115,"cache_read":0.0741,"context":131072,"quantization":"fp8"},{"provider":"Friendli","tag":"friendli","kind":"host","input":0.5,"output":1.5,"cache_read":0.25,"context":163840,"quantization":null},{"provider":"Google","tag":"google-vertex","kind":"cloud","input":0.56,"output":1.68,"cache_read":null,"context":163840,"quantization":null},{"provider":"Phala","tag":"phala","kind":"host","input":1,"output":1,"cache_read":0.5,"context":163840,"quantization":null},{"provider":"Mara","tag":"mara","kind":"host","input":3,"output":4.5,"cache_read":null,"context":32768,"quantization":null},{"provider":"SambaNova","tag":"sambanova","kind":"host","input":3,"output":4.5,"cache_read":null,"context":32768,"quantization":null}]},{"id":"meta-llama/llama-4-maverick","slug":"llama-4-maverick","name":"Llama 4 Maverick","vendor":"meta-llama","vendor_name":"Meta","provider":"OpenRouter","provider_id":"openrouter","family":"llama","input":0.1875,"output":0.6525,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":16384,"speed":null,"modalities":["text","image"],"features":["json","tools"],"description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","url":"https://openrouter.ai/meta-llama/llama-4-maverick","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":0.1875,"output":0.6525,"cache_read":null,"context":128000,"quantization":null},{"provider":"DeepInfra","tag":"deepinfra/base","kind":"host","input":0.2,"output":0.8,"cache_read":null,"context":1048576,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.27,"output":0.85,"cache_read":null,"context":1048576,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.35,"output":1,"cache_read":0.17,"context":524288,"quantization":"fp8"},{"provider":"Google","tag":"google-vertex/us-east5","kind":"cloud","input":0.35,"output":1.15,"cache_read":null,"context":524288,"quantization":null}]},{"id":"deepseek/deepseek-v3.2-exp","slug":"deepseek-v3-2-exp","name":"DeepSeek V3.2 Exp","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.27,"output":0.41,"context":163840,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":65536,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","url":"https://openrouter.ai/deepseek/deepseek-v3.2-exp","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.27,"output":0.41,"cache_read":null,"context":163840,"quantization":"fp8"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.27,"output":0.41,"cache_read":0.27,"context":163840,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.27,"output":0.41,"cache_read":null,"context":163840,"quantization":"fp8"}]},{"id":"z-ai/glm-4.5-air","slug":"glm-4-5-air","name":"GLM 4.5 Air","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":0.13,"output":0.85,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.025,"max_output":98304,"speed":null,"modalities":["text"],"features":["reasoning","tools"],"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","url":"https://openrouter.ai/z-ai/glm-4.5-air","direct_input":0.2,"direct_output":1.1,"direct_cache_read":0.03,"direct_source":"z-ai","routes":[{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.13,"output":0.85,"cache_read":0.025,"context":131072,"quantization":"bf16"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.14,"output":0.86,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"Z.AI","tag":"z-ai/fp8","kind":"official","input":0.2,"output":1.1,"cache_read":0.03,"context":131072,"quantization":"fp8"}]},{"id":"deepseek/deepseek-v4-flash-vision-exp","slug":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.22,"output":0.66,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.007,"max_output":943718,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","url":"https://openrouter.ai/deepseek/deepseek-v4-flash-vision-exp","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.2156,"output":0.6468,"cache_read":0.0069,"context":1048576,"quantization":"fp8"},{"provider":"Fireworks","tag":"fireworks","kind":"host","input":0.22,"output":0.66,"cache_read":0.007,"context":1048576,"quantization":null},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.44,"output":1.32,"cache_read":0.014,"context":1048575,"quantization":"fp8"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.44,"output":1.32,"cache_read":0.028,"context":1048576,"quantization":"fp8"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.44,"output":1.32,"cache_read":0.028,"context":1048576,"quantization":"fp8"},{"provider":"Novita","tag":"novita","kind":"host","input":0.44,"output":1.32,"cache_read":0.028,"context":1048576,"quantization":null}]},{"id":"qwen/qwen3-next-80b-a3b-instruct","slug":"qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.09,"output":1.1,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":16384,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","url":"https://openrouter.ai/qwen/qwen3-next-80b-a3b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.0975,"output":0.78,"cache_read":null,"context":131072,"quantization":null},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.09,"output":1.1,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.1,"output":1.1,"cache_read":0.07,"context":262144,"quantization":"fp8"},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":0.15,"output":1.2,"cache_read":null,"context":262144,"quantization":null},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.15,"output":1.5,"cache_read":null,"context":131072,"quantization":"bf16"}]},{"id":"qwen/qwen-2.5-72b-instruct","slug":"qwen-2-5-72b-instruct","name":"Qwen2.5 72B Instruct","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.36,"output":0.4,"context":32768,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":16384,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","url":"https://openrouter.ai/qwen/qwen-2.5-72b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.36,"output":0.4,"cache_read":null,"context":32768,"quantization":"fp8"},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.38,"output":0.4,"cache_read":null,"context":32000,"quantization":"bf16"}]},{"id":"qwen/qwen-plus-2025-07-28","slug":"qwen-plus-2025-07-28","name":"Qwen Plus 0728","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.26,"output":0.78,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":32768,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","url":"https://openrouter.ai/qwen/qwen-plus-2025-07-28","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.26,"output":0.78,"cache_read":null,"context":1000000,"quantization":null}]},{"id":"qwen/qwen-plus","slug":"qwen-plus","name":"Qwen-Plus","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.26,"output":0.78,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.052,"max_output":32768,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","url":"https://openrouter.ai/qwen/qwen-plus","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.26,"output":0.78,"cache_read":0.052,"context":1000000,"quantization":null}]},{"id":"qwen/qwen3-coder-flash","slug":"qwen3-coder-flash","name":"Qwen3 Coder Flash","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.195,"output":0.975,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.039,"max_output":65536,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","url":"https://openrouter.ai/qwen/qwen3-coder-flash","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.195,"output":0.975,"cache_read":0.039,"context":1000000,"quantization":null}]},{"id":"meta-llama/llama-3.1-70b-instruct","slug":"llama-3-1-70b-instruct","name":"Llama 3.1 70B Instruct","vendor":"meta-llama","vendor_name":"Meta","provider":"OpenRouter","provider_id":"openrouter","family":"llama","input":0.4,"output":0.4,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":16384,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","url":"https://openrouter.ai/meta-llama/llama-3.1-70b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/turbo","kind":"host","input":0.4,"output":0.4,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":0.72,"output":0.72,"cache_read":null,"context":131072,"quantization":null}]},{"id":"mistralai/mistral-small-3.1-24b-instruct","slug":"mistral-small-3-1-24b-instruct","name":"Mistral Small 3.1 24B","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.351,"output":0.555,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":102400,"speed":null,"modalities":["text","image"],"features":[],"description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","url":"https://openrouter.ai/mistralai/mistral-small-3.1-24b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.351,"output":0.555,"cache_read":null,"context":128000,"quantization":null}]},{"id":"qwen/qwen3-next-80b-a3b-thinking","slug":"qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.15,"output":1.2,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":32768,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","url":"https://openrouter.ai/qwen/qwen3-next-80b-a3b-thinking","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":0.15,"output":1.2,"cache_read":null,"context":262144,"quantization":null},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.15,"output":1.2,"cache_read":null,"context":131072,"quantization":null}]},{"id":"qwen/qwen3.6-flash","slug":"qwen3-6-flash","name":"Qwen3.6 Flash","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.1875,"output":1.125,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":65536,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","url":"https://openrouter.ai/qwen/qwen3.6-flash","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.1875,"output":1.125,"cache_read":null,"context":1000000,"quantization":null}]},{"id":"deepseek/deepseek-chat-v3.1","slug":"deepseek-chat-v3-1","name":"DeepSeek V3.1","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.25,"output":0.95,"context":163840,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.13,"max_output":32768,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","url":"https://openrouter.ai/deepseek/deepseek-chat-v3.1","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.25,"output":0.95,"cache_read":0.13,"context":163840,"quantization":"fp4"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.27,"output":1,"cache_read":null,"context":163840,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.27,"output":1,"cache_read":0.135,"context":131072,"quantization":"fp8"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.3,"output":0.95,"cache_read":0.13,"context":131072,"quantization":"fp8"},{"provider":"CoreWeave","tag":"coreweave/fp8","kind":"host","input":0.55,"output":1.65,"cache_read":0.55,"context":161000,"quantization":"fp8"},{"provider":"SambaNova","tag":"sambanova/fp8","kind":"host","input":0.65,"output":1.5,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"Mara","tag":"mara","kind":"host","input":0.6,"output":1.7,"cache_read":null,"context":131072,"quantization":null},{"provider":"Google","tag":"google-vertex/us-west2","kind":"cloud","input":0.6,"output":1.7,"cache_read":null,"context":163840,"quantization":null}]},{"id":"deepseek/deepseek-chat-v3-0324","slug":"deepseek-chat-v3-0324","name":"DeepSeek V3 0324","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.25,"output":1,"context":163840,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":147456,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","url":"https://openrouter.ai/deepseek/deepseek-chat-v3-0324","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.24,"output":0.9,"cache_read":0.135,"context":163840,"quantization":"fp4"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.25,"output":1,"cache_read":null,"context":163840,"quantization":"fp8"},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.29,"output":1.14,"cache_read":0.11,"context":163840,"quantization":"fp8"}]},{"id":"qwen/qwen3.5-35b-a3b","slug":"qwen3-5-35b-a3b","name":"Qwen3.5-35B-A3B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.1625,"output":1.3,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":65536,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","url":"https://openrouter.ai/qwen/qwen3.5-35b-a3b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Darkbloom","tag":"darkbloom/fp4","kind":"host","input":0.08,"output":0.75,"cache_read":null,"context":262144,"quantization":"fp4"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.14,"output":1,"cache_read":0.05,"context":262144,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.15,"output":1,"cache_read":0.05,"context":262144,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.1625,"output":1.3,"cache_read":null,"context":262144,"quantization":null},{"provider":"Venice","tag":"venice","kind":"host","input":0.3125,"output":1.25,"cache_read":0.1562,"context":256000,"quantization":null},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.225,"output":1.8,"cache_read":0.225,"context":262144,"quantization":"fp8"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.24,"output":1.8,"cache_read":0.15,"context":262144,"quantization":"fp8"}]},{"id":"mistralai/codestral-2508","slug":"codestral-2508","name":"Codestral 2508","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.3,"output":0.9,"context":256000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.03,"max_output":204800,"speed":null,"modalities":["text","file"],"features":["json","tools"],"description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation. [Blog Post](https://mistral.ai/news/codestral-25-08)","url":"https://openrouter.ai/mistralai/codestral-2508","direct_input":0.3,"direct_output":0.9,"direct_cache_read":0.03,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral","kind":"official","input":0.3,"output":0.9,"cache_read":0.03,"context":256000,"quantization":null}]},{"id":"z-ai/glm-4.6v","slug":"glm-4-6v","name":"GLM 4.6V","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":0.3,"output":0.9,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.055,"max_output":32768,"speed":null,"modalities":["image","text","video"],"features":["json","reasoning","tools"],"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","url":"https://openrouter.ai/z-ai/glm-4.6v","direct_input":0.3,"direct_output":0.9,"direct_cache_read":0.05,"direct_source":"z-ai","routes":[{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.3,"output":0.9,"cache_read":0.055,"context":131072,"quantization":"bf16"},{"provider":"Z.AI","tag":"z-ai/fp8","kind":"official","input":0.3,"output":0.9,"cache_read":0.05,"context":131072,"quantization":"fp8"}]},{"id":"openai/gpt-5.6-luna","slug":"gpt-5-6-luna","name":"GPT-5.6 Luna","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.2,"output":1.2,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.02,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","url":"https://openrouter.ai/openai/gpt-5.6-luna","direct_input":0.1,"direct_output":0.6,"direct_cache_read":0.01,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":0.1,"output":0.6,"cache_read":0.01,"context":1050000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":0.2,"output":1.2,"cache_read":0.02,"context":1050000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/us-east-1","kind":"cloud","input":0.22,"output":1.32,"cache_read":0.022,"context":1050000,"quantization":null}]},{"id":"openai/gpt-5.6-luna-pro","slug":"gpt-5-6-luna-pro","name":"GPT-5.6 Luna Pro","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.2,"output":1.2,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.02,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","url":"https://openrouter.ai/openai/gpt-5.6-luna-pro","direct_input":0.1,"direct_output":0.6,"direct_cache_read":0.01,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":0.1,"output":0.6,"cache_read":0.01,"context":1050000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":0.2,"output":1.2,"cache_read":0.02,"context":1050000,"quantization":null}]},{"id":"deepseek/deepseek-chat","slug":"deepseek-chat","name":"DeepSeek V3","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.2574,"output":1.0287,"context":163840,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":16000,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","url":"https://openrouter.ai/deepseek/deepseek-chat","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"StreamLake","tag":"streamlake","kind":"host","input":0.2574,"output":1.0287,"cache_read":null,"context":128000,"quantization":null},{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.32,"output":0.89,"cache_read":null,"context":163840,"quantization":"fp4"}]},{"id":"deepseek/deepseek-v3.1-terminus","slug":"deepseek-v3-1-terminus","name":"DeepSeek V3.1 Terminus","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.27,"output":1,"context":163840,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.135,"max_output":32768,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","url":"https://openrouter.ai/deepseek/deepseek-v3.1-terminus","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.27,"output":1,"cache_read":null,"context":163840,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.27,"output":1,"cache_read":0.135,"context":131072,"quantization":"fp8"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.3,"output":0.95,"cache_read":0.13,"context":131072,"quantization":"fp8"},{"provider":"StreamLake","tag":"streamlake","kind":"host","input":0.3426,"output":1.0284,"cache_read":null,"context":128000,"quantization":null}]},{"id":"openai/gpt-5.4-nano","slug":"gpt-5-4-nano","name":"GPT-5.4 Nano","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.2,"output":1.25,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.02,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","url":"https://openrouter.ai/openai/gpt-5.4-nano","direct_input":0.1,"direct_output":0.625,"direct_cache_read":0.01,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":0.1,"output":0.625,"cache_read":0.01,"context":400000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":0.2,"output":1.25,"cache_read":0.02,"context":400000,"quantization":null}]},{"id":"qwen/qwen3-coder","slug":"qwen3-coder","name":"Qwen3 Coder 480B A35B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.3,"output":1,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.1,"max_output":65536,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","url":"https://openrouter.ai/qwen/qwen3-coder","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/turbo","kind":"host","input":0.3,"output":1,"cache_read":0.1,"context":262144,"quantization":"fp4"},{"provider":"Google","tag":"google-vertex/us-south1","kind":"cloud","input":0.22,"output":1.8,"cache_read":null,"context":262144,"quantization":null},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.35,"output":1.5,"cache_read":0.04,"context":256000,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.38,"output":1.55,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba/opensource","kind":"host","input":0.975,"output":4.875,"cache_read":null,"context":262144,"quantization":null}]},{"id":"anthropic/claude-3-haiku","slug":"claude-3-haiku","name":"Claude 3 Haiku","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":0.25,"output":1.25,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.03,"max_output":4096,"speed":null,"modalities":["text","image"],"features":["tools"],"description":"Claude 3 Haiku is Anthropic's fastest and most compact model for near-instant responsiveness. Quick and accurate targeted performance. See the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku) #multimodal","url":"https://openrouter.ai/anthropic/claude-3-haiku","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":0.25,"output":1.25,"cache_read":0.03,"context":200000,"quantization":null}]},{"id":"deepseek/deepseek-v4.1-flash","slug":"deepseek-v4-1-flash","name":"DeepSeek V4.1 Flash","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.3,"output":1.2,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.006,"max_output":384000,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","url":"https://openrouter.ai/deepseek/deepseek-v4.1-flash","direct_input":0.3,"direct_output":1.2,"direct_cache_read":0.006,"direct_source":"deepseek","routes":[{"provider":"Relace","tag":"relace/fp4","kind":"host","input":0.15,"output":0.6,"cache_read":0.015,"context":1048576,"quantization":"fp4"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.2,"output":0.6,"cache_read":0.006,"context":1048576,"quantization":"fp8"},{"provider":"Fireworks","tag":"fireworks","kind":"host","input":0.22,"output":0.66,"cache_read":0.007,"context":1048576,"quantization":null},{"provider":"Morph","tag":"morph/fp8","kind":"host","input":0.21,"output":0.84,"cache_read":0.021,"context":1048576,"quantization":"fp8"},{"provider":"Reka","tag":"reka/fp4","kind":"host","input":0.29,"output":1.16,"cache_read":0.029,"context":262144,"quantization":"fp4"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.3,"output":1.2,"cache_read":0.03,"context":1000000,"quantization":null},{"provider":"Together","tag":"together","kind":"host","input":0.3,"output":1.2,"cache_read":0.006,"context":1048576,"quantization":null},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.3,"output":1.2,"cache_read":0.006,"context":1048576,"quantization":"fp8"},{"provider":"Modal","tag":"modal","kind":"host","input":0.3,"output":1.2,"cache_read":0.03,"context":1048576,"quantization":null},{"provider":"Wafer","tag":"wafer","kind":"host","input":0.3,"output":1.2,"cache_read":0.006,"context":1048576,"quantization":null},{"provider":"BaseTen","tag":"baseten/fp8","kind":"host","input":0.3,"output":1.2,"cache_read":0.03,"context":1048576,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.3,"output":1.2,"cache_read":0.006,"context":1048576,"quantization":"fp8"},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.3,"output":1.2,"cache_read":0.006,"context":1048575,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.3,"output":1.2,"cache_read":0.006,"context":1048576,"quantization":"fp8"},{"provider":"DeepSeek","tag":"deepseek","kind":"official","input":0.3,"output":1.2,"cache_read":0.006,"context":1048576,"quantization":null},{"provider":"Phala","tag":"phala","kind":"host","input":0.345,"output":1.38,"cache_read":0.0069,"context":1048576,"quantization":null},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.375,"output":1.5,"cache_read":0.0075,"context":1000000,"quantization":"fp8"}]},{"id":"qwen/qwen3.5-27b","slug":"qwen3-5-27b","name":"Qwen3.5-27B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.195,"output":1.56,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":65536,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","url":"https://openrouter.ai/qwen/qwen3.5-27b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.195,"output":1.56,"cache_read":null,"context":262144,"quantization":null},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.25,"output":2,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.27,"output":2.16,"cache_read":0.27,"context":262144,"quantization":"fp8"},{"provider":"Phala","tag":"phala","kind":"host","input":0.3,"output":2.4,"cache_read":0.15,"context":262144,"quantization":null},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.3,"output":2.4,"cache_read":null,"context":262144,"quantization":"bf16"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.26,"output":2.6,"cache_read":null,"context":262144,"quantization":"fp8"}]},{"id":"qwen/qwen3.7-plus","slug":"qwen3-7-plus","name":"Qwen3.7 Plus","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.32,"output":1.28,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.064,"max_output":131072,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","url":"https://openrouter.ai/qwen/qwen3.7-plus","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.32,"output":1.28,"cache_read":0.064,"context":1000000,"quantization":null}]},{"id":"google/gemini-3.1-flash-lite","slug":"gemini-3-1-flash-lite","name":"Gemini 3.1 Flash Lite","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":0.25,"output":1.5,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.025,"max_output":65536,"speed":null,"modalities":["text","image","video","file","audio"],"features":["json","reasoning","tools"],"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","url":"https://openrouter.ai/google/gemini-3.1-flash-lite","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google","tag":"google-vertex/global/flex","kind":"cloud","input":0.125,"output":0.75,"cache_read":0.0125,"context":1048576,"quantization":null},{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.125,"output":0.75,"cache_read":0.0125,"context":1048576,"quantization":null}]},{"id":"google/gemini-3.1-flash-lite-preview","slug":"gemini-3-1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":0.25,"output":1.5,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.025,"max_output":65536,"speed":null,"modalities":["text","image","video","file","audio"],"features":["json","reasoning","tools"],"description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","url":"https://openrouter.ai/google/gemini-3.1-flash-lite-preview","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.125,"output":0.75,"cache_read":0.0125,"context":1048576,"quantization":null}]},{"id":"qwen/qwen3.5-plus-02-15","slug":"qwen3-5-plus-02-15","name":"Qwen3.5 Plus 2026-02-15","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.26,"output":1.56,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":65536,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","url":"https://openrouter.ai/qwen/qwen3.5-plus-02-15","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.26,"output":1.56,"cache_read":null,"context":1000000,"quantization":null}]},{"id":"qwen/qwen3-vl-235b-a22b-instruct","slug":"qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.21,"output":1.9,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.1,"max_output":32768,"speed":null,"modalities":["text","image"],"features":["json","tools"],"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","url":"https://openrouter.ai/qwen/qwen3-vl-235b-a22b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.2,"output":0.88,"cache_read":0.11,"context":262144,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.26,"output":1.04,"cache_read":null,"context":131072,"quantization":null},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.3,"output":1.5,"cache_read":null,"context":131072,"quantization":"bf16"},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.21,"output":1.9,"cache_read":0.1,"context":128000,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.21,"output":1.9,"cache_read":0.1,"context":131072,"quantization":"fp8"}]},{"id":"google/gemma-2-27b-it","slug":"gemma-2-27b-it","name":"Gemma 2 27B","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemma","input":0.65,"output":0.65,"context":8192,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":2048,"speed":null,"modalities":["text"],"features":["json"],"description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","url":"https://openrouter.ai/google/gemma-2-27b-it","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"NextBit","tag":"nextbit/int4","kind":"host","input":0.65,"output":0.65,"cache_read":null,"context":8192,"quantization":"int4"}]},{"id":"qwen/qwen3-vl-8b-thinking","slug":"qwen3-vl-8b-thinking","name":"Qwen3 VL 8B Thinking","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.18,"output":2.1,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":32768,"speed":null,"modalities":["image","text"],"features":["json","reasoning","tools"],"description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","url":"https://openrouter.ai/qwen/qwen3-vl-8b-thinking","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.18,"output":2.1,"cache_read":null,"context":131072,"quantization":null}]},{"id":"qwen/qwen3.5-plus-20260420","slug":"qwen3-5-plus-20260420","name":"Qwen3.5 Plus 2026-04-20","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.3,"output":1.8,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":65536,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","url":"https://openrouter.ai/qwen/qwen3.5-plus-20260420","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.3,"output":1.8,"cache_read":null,"context":1000000,"quantization":null}]},{"id":"openai/gpt-5-mini","slug":"gpt-5-mini","name":"GPT-5 Mini","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.25,"output":2,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.025,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","url":"https://openrouter.ai/openai/gpt-5-mini","direct_input":0.125,"direct_output":1,"direct_cache_read":0.0125,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":0.125,"output":1,"cache_read":0.0125,"context":400000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":0.25,"output":2,"cache_read":0.03,"context":400000,"quantization":null}]},{"id":"openai/gpt-5.1-codex-mini","slug":"gpt-5-1-codex-mini","name":"GPT-5.1-Codex-Mini","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.25,"output":2,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.03,"max_output":128000,"speed":null,"modalities":["image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","url":"https://openrouter.ai/openai/gpt-5.1-codex-mini","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":0.25,"output":2,"cache_read":0.03,"context":400000,"quantization":null}]},{"id":"openai/gpt-4.1-mini","slug":"gpt-4-1-mini","name":"GPT-4.1 Mini","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.4,"output":1.6,"context":1047576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.1,"max_output":32768,"speed":null,"modalities":["image","text","file"],"features":["json","tools"],"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","url":"https://openrouter.ai/openai/gpt-4.1-mini","direct_input":0.4,"direct_output":1.6,"direct_cache_read":0.1,"direct_source":"openai","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":0.4,"output":1.6,"cache_read":0.1,"context":1047576,"quantization":null},{"provider":"OpenAI","tag":"openai","kind":"official","input":0.4,"output":1.6,"cache_read":0.1,"context":1047576,"quantization":null}]},{"id":"qwen/qwen3.5-122b-a10b","slug":"qwen3-5-122b-a10b","name":"Qwen3.5-122B-A10B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.26,"output":2.08,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":65536,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","url":"https://openrouter.ai/qwen/qwen3.5-122b-a10b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.26,"output":2.08,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.26,"output":2.08,"cache_read":null,"context":262144,"quantization":null},{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.29,"output":2.4,"cache_read":null,"context":262144,"quantization":"fp4"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.3,"output":2.4,"cache_read":0.3,"context":262144,"quantization":"fp8"},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.4,"output":3.2,"cache_read":null,"context":262144,"quantization":"bf16"}]},{"id":"qwen/qwen3.6-27b","slug":"qwen3-6-27b","name":"Qwen3.6 27B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.3,"output":2,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.03,"max_output":65536,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","url":"https://openrouter.ai/qwen/qwen3.6-27b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Chutes","tag":"chutes/fp8","kind":"host","input":0.3,"output":2,"cache_read":0.03,"context":262144,"quantization":"fp8"},{"provider":"Phala","tag":"phala","kind":"host","input":0.32,"output":2.7,"cache_read":0.15,"context":262144,"quantization":null},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.45,"output":2.7,"cache_read":null,"context":262144,"quantization":null},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.3,"output":3.2,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.32,"output":3.2,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.325,"output":3.25,"cache_read":null,"context":256000,"quantization":"fp8"}]},{"id":"qwen/qwen3.6-plus","slug":"qwen3-6-plus","name":"Qwen3.6 Plus","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.325,"output":1.95,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":65536,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","url":"https://openrouter.ai/qwen/qwen3.6-plus","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.325,"output":1.95,"cache_read":null,"context":1000000,"quantization":null}]},{"id":"z-ai/glm-4.7","slug":"glm-4-7","name":"GLM 4.7","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":0.4,"output":1.75,"context":204800,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.08,"max_output":131072,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","url":"https://openrouter.ai/z-ai/glm-4.7","direct_input":0.6,"direct_output":2.2,"direct_cache_read":0.11,"direct_source":"z-ai","routes":[{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.4,"output":1.75,"cache_read":0.08,"context":202752,"quantization":"fp4"},{"provider":"Venice","tag":"venice/fp4","kind":"host","input":0.4004,"output":1.9292,"cache_read":0.0801,"context":198000,"quantization":"fp4"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.52,"output":1.85,"cache_read":0.12,"context":202752,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.54,"output":1.98,"cache_read":0.099,"context":204800,"quantization":"fp8"},{"provider":"Google","tag":"google-vertex","kind":"cloud","input":0.6,"output":2.2,"cache_read":null,"context":200000,"quantization":null},{"provider":"Z.AI","tag":"z-ai/fp4","kind":"official","input":0.6,"output":2.2,"cache_read":0.11,"context":202752,"quantization":"fp4"},{"provider":"Mancer 2","tag":"mancer/fp4","kind":"host","input":0.7,"output":2.5,"cache_read":null,"context":131072,"quantization":"fp4"}]},{"id":"qwen/qwen-2.5-coder-32b-instruct","slug":"qwen-2-5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.66,"output":1,"context":32768,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":29491,"speed":null,"modalities":["text"],"features":[],"description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","url":"https://openrouter.ai/qwen/qwen-2.5-coder-32b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.66,"output":1,"cache_read":null,"context":32768,"quantization":null}]},{"id":"qwen/qwen3-235b-a22b-thinking-2507","slug":"qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking 2507","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.23,"output":2.3,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":117964,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","url":"https://openrouter.ai/qwen/qwen3-235b-a22b-thinking-2507","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.23,"output":2.3,"cache_read":null,"context":131072,"quantization":null},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.3,"output":3,"cache_read":null,"context":131072,"quantization":"fp8"},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.45,"output":3.5,"cache_read":null,"context":128000,"quantization":"fp8"}]},{"id":"mistralai/mistral-large-2512","slug":"mistral-large-2512","name":"Mistral Large 3 2512","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.5,"output":1.5,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.05,"max_output":209715,"speed":null,"modalities":["text","image","file"],"features":["json","tools"],"description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","url":"https://openrouter.ai/mistralai/mistral-large-2512","direct_input":0.5,"direct_output":1.5,"direct_cache_read":0.05,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/zdr","kind":"official","input":0.5,"output":1.5,"cache_read":0.05,"context":262144,"quantization":null}]},{"id":"openai/gpt-3.5-turbo","slug":"gpt-3-5-turbo","name":"GPT-3.5 Turbo","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.5,"output":1.5,"context":16385,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":4096,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks. Training data up to Sep 2021.","url":"https://openrouter.ai/openai/gpt-3.5-turbo","direct_input":0.5,"direct_output":1.5,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":0.5,"output":1.5,"cache_read":null,"context":16385,"quantization":null}]},{"id":"qwen/qwen3-30b-a3b-thinking-2507","slug":"qwen3-30b-a3b-thinking-2507","name":"Qwen3 30B A3B Thinking 2507","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.2,"output":2.4,"context":81920,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":32768,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","url":"https://openrouter.ai/qwen/qwen3-30b-a3b-thinking-2507","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.2,"output":2.4,"cache_read":null,"context":81920,"quantization":null}]},{"id":"qwen/qwen3-vl-30b-a3b-thinking","slug":"qwen3-vl-30b-a3b-thinking","name":"Qwen3 VL 30B A3B Thinking","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.2,"output":2.4,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":32768,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","url":"https://openrouter.ai/qwen/qwen3-vl-30b-a3b-thinking","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.29,"output":1,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.2,"output":2.4,"cache_read":null,"context":131072,"quantization":null}]},{"id":"z-ai/glm-4.6","slug":"glm-4-6","name":"GLM 4.6","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":0.43,"output":1.75,"context":204800,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.08,"max_output":16384,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","url":"https://openrouter.ai/z-ai/glm-4.6","direct_input":0.6,"direct_output":2.2,"direct_cache_read":0.11,"direct_source":"z-ai","routes":[{"provider":"Venice","tag":"venice/fp4","kind":"host","input":0.43,"output":1.75,"cache_read":0.08,"context":198000,"quantization":"fp4"},{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.5,"output":2,"cache_read":0.1,"context":202752,"quantization":"fp4"},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.55,"output":2.2,"cache_read":0.11,"context":204800,"quantization":"bf16"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.6,"output":2.2,"cache_read":0.11,"context":202752,"quantization":"fp8"},{"provider":"Z.AI","tag":"z-ai/fp4","kind":"official","input":0.6,"output":2.2,"cache_read":0.11,"context":202752,"quantization":"fp4"}]},{"id":"qwen/qwen3-235b-a22b","slug":"qwen3-235b-a22b","name":"Qwen3 235B A22B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.455,"output":1.82,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":8192,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","url":"https://openrouter.ai/qwen/qwen3-235b-a22b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.455,"output":1.82,"cache_read":null,"context":131072,"quantization":null}]},{"id":"qwen/qwen3.8-27b","slug":"qwen3-8-27b","name":"Qwen3.8 27B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.214,"output":2.55,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.15,"max_output":131072,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","url":"https://openrouter.ai/qwen/qwen3.8-27b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/bf16","kind":"host","input":0.15,"output":1.875,"cache_read":0.0375,"context":262144,"quantization":"bf16"},{"provider":"Darkbloom","tag":"darkbloom/fp4","kind":"host","input":0.15,"output":2,"cache_read":null,"context":262144,"quantization":"fp4"},{"provider":"Phala","tag":"phala","kind":"host","input":0.24,"output":2.2,"cache_read":0.05,"context":262144,"quantization":null},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.24,"output":2.2,"cache_read":0.05,"context":262144,"quantization":"fp8"},{"provider":"AkashML","tag":"akashml/fp8","kind":"host","input":0.25,"output":2.2,"cache_read":0.05,"context":262144,"quantization":"fp8"},{"provider":"DekaLLM","tag":"dekallm","kind":"host","input":0.2,"output":2.5,"cache_read":0.05,"context":262144,"quantization":null},{"provider":"Reka","tag":"reka/fp8","kind":"host","input":0.214,"output":2.55,"cache_read":0.15,"context":262144,"quantization":"fp8"},{"provider":"Chutes","tag":"chutes/fp8","kind":"host","input":0.32,"output":2.5,"cache_read":0.032,"context":262144,"quantization":"fp8"},{"provider":"Mancer 2","tag":"mancer/fp8","kind":"host","input":0.25,"output":2.75,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Ionstream","tag":"ionstream/fp8","kind":"host","input":0.35,"output":2.55,"cache_read":0.05,"context":262144,"quantization":"fp8"},{"provider":"Io Net","tag":"io-net/fp8","kind":"host","input":0.3,"output":2.8,"cache_read":0.18,"context":65536,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.425,"output":2.55,"cache_read":0.085,"context":1000000,"quantization":null},{"provider":"CoreWeave","tag":"coreweave/fp8","kind":"host","input":0.4,"output":3,"cache_read":0.15,"context":262144,"quantization":"fp8"},{"provider":"Novita","tag":"novita","kind":"host","input":0.42,"output":3,"cache_read":0.085,"context":1000000,"quantization":null},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.45,"output":3.2,"cache_read":0.05,"context":262144,"quantization":null},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":0.45,"output":3.2,"cache_read":null,"context":262144,"quantization":"fp8"}]},{"id":"deepseek/deepseek-r1-distill-llama-70b","slug":"deepseek-r1-distill-llama-70b","name":"R1 Distill Llama 70B","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"llama","input":0.8,"output":0.8,"context":8192,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":7372,"speed":null,"modalities":["text"],"features":["reasoning"],"description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","url":"https://openrouter.ai/deepseek/deepseek-r1-distill-llama-70b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.8,"output":0.8,"cache_read":null,"context":8192,"quantization":"bf16"}]},{"id":"mistralai/devstral-2512","slug":"devstral-2512","name":"Devstral 2 2512","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.4,"output":2,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.04,"max_output":209715,"speed":null,"modalities":["text","file"],"features":["json","tools"],"description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","url":"https://openrouter.ai/mistralai/devstral-2512","direct_input":0.4,"direct_output":2,"direct_cache_read":0.04,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/eu","kind":"official","input":0.4,"output":2,"cache_read":0.04,"context":262144,"quantization":null}]},{"id":"mistralai/mistral-medium-3","slug":"mistral-medium-3","name":"Mistral Medium 3","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.4,"output":2,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.04,"max_output":104857,"speed":null,"modalities":["text","image","file"],"features":["json","tools"],"description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","url":"https://openrouter.ai/mistralai/mistral-medium-3","direct_input":0.4,"direct_output":2,"direct_cache_read":0.04,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/zdr","kind":"official","input":0.4,"output":2,"cache_read":0.04,"context":131072,"quantization":null}]},{"id":"mistralai/mistral-medium-3.1","slug":"mistral-medium-3-1","name":"Mistral Medium 3.1","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":0.4,"output":2,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.04,"max_output":104857,"speed":null,"modalities":["text","image","file"],"features":["json","tools"],"description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","url":"https://openrouter.ai/mistralai/mistral-medium-3.1","direct_input":0.4,"direct_output":2,"direct_cache_read":0.04,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/zdr","kind":"official","input":0.4,"output":2,"cache_read":0.04,"context":131072,"quantization":null}]},{"id":"amazon/nova-2-lite-v1","slug":"nova-2-lite-v1","name":"Nova 2 Lite","vendor":"amazon","vendor_name":"Amazon","provider":"OpenRouter","provider_id":"openrouter","family":"nova","input":0.3,"output":2.5,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":65535,"speed":null,"modalities":["text","image","video","file"],"features":["reasoning","tools"],"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","url":"https://openrouter.ai/amazon/nova-2-lite-v1","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Amazon Bedrock","tag":"amazon-bedrock/eu-west-1","kind":"cloud","input":0.3,"output":2.5,"cache_read":null,"context":1000000,"quantization":null}]},{"id":"google/gemini-2.5-flash","slug":"gemini-2-5-flash","name":"Gemini 2.5 Flash","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":0.3,"output":2.5,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.03,"max_output":65535,"speed":null,"modalities":["file","image","text","audio","video"],"features":["json","reasoning","tools"],"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","url":"https://openrouter.ai/google/gemini-2.5-flash","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.15,"output":1.25,"cache_read":0.015,"context":1048576,"quantization":null},{"provider":"Google","tag":"google-vertex/eu","kind":"cloud","input":0.3,"output":2.5,"cache_read":0.03,"context":1048576,"quantization":null}]},{"id":"google/gemini-3.5-flash-lite","slug":"gemini-3-5-flash-lite","name":"Gemini 3.5 Flash Lite","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":0.3,"output":2.5,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.03,"max_output":65536,"speed":null,"modalities":["text","image","video","file","audio"],"features":["json","reasoning","tools"],"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","url":"https://openrouter.ai/google/gemini-3.5-flash-lite","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google","tag":"google-vertex/global/flex","kind":"cloud","input":0.15,"output":1.25,"cache_read":0.015,"context":1048576,"quantization":null},{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.15,"output":1.25,"cache_read":0.015,"context":1048576,"quantization":null}]},{"id":"qwen/qwen2.5-vl-72b-instruct","slug":"qwen2-5-vl-72b-instruct","name":"Qwen2.5 VL 72B Instruct","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.8,"output":1,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.4,"max_output":115200,"speed":null,"modalities":["text","image"],"features":["json"],"description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","url":"https://openrouter.ai/qwen/qwen2.5-vl-72b-instruct","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.8,"output":1,"cache_read":0.4,"context":128000,"quantization":"fp8"}]},{"id":"z-ai/glm-4.5v","slug":"glm-4-5v","name":"GLM 4.5V","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":0.6,"output":1.8,"context":65536,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.11,"max_output":16384,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","url":"https://openrouter.ai/z-ai/glm-4.5v","direct_input":0.6,"direct_output":1.8,"direct_cache_read":0.11,"direct_source":"z-ai","routes":[{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.6,"output":1.8,"cache_read":0.11,"context":65536,"quantization":"fp8"},{"provider":"Z.AI","tag":"z-ai/fp8","kind":"official","input":0.6,"output":1.8,"cache_read":0.11,"context":65536,"quantization":"fp8"}]},{"id":"moonshotai/kimi-k2.5","slug":"kimi-k2-5","name":"Kimi K2.5","vendor":"moonshotai","vendor_name":"Moonshot","provider":"OpenRouter","provider_id":"openrouter","family":"kimi","input":0.45,"output":2.25,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.07,"max_output":235929,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","url":"https://openrouter.ai/moonshotai/kimi-k2.5","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"SiliconFlow","tag":"siliconflow/int4","kind":"host","input":0.45,"output":2.25,"cache_read":0.07,"context":262144,"quantization":"int4"},{"provider":"AtlasCloud","tag":"atlas-cloud/int4","kind":"host","input":0.49,"output":2.5,"cache_read":0.2,"context":262144,"quantization":"int4"},{"provider":"Novita","tag":"novita","kind":"host","input":0.57,"output":2.85,"cache_read":0.095,"context":262144,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/us-east-2","kind":"cloud","input":0.6,"output":3,"cache_read":null,"context":262144,"quantization":null},{"provider":"Phala","tag":"phala","kind":"host","input":0.6,"output":3,"cache_read":0.22,"context":262144,"quantization":null},{"provider":"Venice","tag":"venice","kind":"host","input":0.532,"output":3.325,"cache_read":0.209,"context":256000,"quantization":null}]},{"id":"deepseek/deepseek-r1-0528","slug":"deepseek-r1-0528","name":"R1 0528","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.5,"output":2.15,"context":163840,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.35,"max_output":32768,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","url":"https://openrouter.ai/deepseek/deepseek-r1-0528","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.5,"output":2.15,"cache_read":0.35,"context":163840,"quantization":"fp4"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.5,"output":2.18,"cache_read":null,"context":163840,"quantization":"fp8"},{"provider":"StreamLake","tag":"streamlake","kind":"host","input":0.571,"output":2.286,"cache_read":null,"context":128000,"quantization":null},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.7,"output":2.5,"cache_read":0.35,"context":163840,"quantization":"fp8"}]},{"id":"z-ai/glm-5","slug":"glm-5","name":"GLM 5","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":0.6,"output":1.92,"context":204800,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.12,"max_output":128000,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","url":"https://openrouter.ai/z-ai/glm-5","direct_input":1,"direct_output":3.2,"direct_cache_read":0.2,"direct_source":"z-ai","routes":[{"provider":"StreamLake","tag":"streamlake/fp8","kind":"host","input":0.6,"output":1.92,"cache_read":0.12,"context":198000,"quantization":"fp8"},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.6,"output":1.92,"cache_read":0.12,"context":202752,"quantization":"fp8"},{"provider":"Baidu","tag":"baidu/fp8","kind":"host","input":0.7,"output":2.24,"cache_read":0.14,"context":202752,"quantization":"fp8"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.95,"output":2.55,"cache_read":0.2,"context":204800,"quantization":"fp8"},{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":1,"output":3.2,"cache_read":null,"context":202752,"quantization":null},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":1,"output":3.2,"cache_read":0.2,"context":198000,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":1,"output":3.2,"cache_read":0.2,"context":202800,"quantization":"fp8"},{"provider":"Z.AI","tag":"z-ai/fp8","kind":"official","input":1,"output":3.2,"cache_read":0.2,"context":202752,"quantization":"fp8"}]},{"id":"z-ai/glm-4.5","slug":"glm-4-5","name":"GLM 4.5","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":0.6,"output":2.2,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.11,"max_output":98304,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","url":"https://openrouter.ai/z-ai/glm-4.5","direct_input":0.6,"direct_output":2.2,"direct_cache_read":0.11,"direct_source":"z-ai","routes":[{"provider":"Z.AI","tag":"z-ai/fp8","kind":"official","input":0.6,"output":2.2,"cache_read":0.11,"context":131072,"quantization":"fp8"}]},{"id":"moonshotai/kimi-k2","slug":"kimi-k2","name":"Kimi K2 0711","vendor":"moonshotai","vendor_name":"Moonshot","provider":"OpenRouter","provider_id":"openrouter","family":"kimi","input":0.57,"output":2.3,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":98304,"speed":null,"modalities":["text"],"features":["tools"],"description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","url":"https://openrouter.ai/moonshotai/kimi-k2","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.57,"output":2.3,"cache_read":null,"context":131072,"quantization":"fp8"}]},{"id":"moonshotai/kimi-k2-0905","slug":"kimi-k2-0905","name":"Kimi K2 0905","vendor":"moonshotai","vendor_name":"Moonshot","provider":"OpenRouter","provider_id":"openrouter","family":"kimi","input":0.6,"output":2.5,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":98304,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","url":"https://openrouter.ai/moonshotai/kimi-k2-0905","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.6,"output":2.5,"cache_read":null,"context":262144,"quantization":"fp8"}]},{"id":"moonshotai/kimi-k2-thinking","slug":"kimi-k2-thinking","name":"Kimi K2 Thinking","vendor":"moonshotai","vendor_name":"Moonshot","provider":"OpenRouter","provider_id":"openrouter","family":"kimi","input":0.6,"output":2.5,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.15,"max_output":98304,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","url":"https://openrouter.ai/moonshotai/kimi-k2-thinking","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google","tag":"google-vertex","kind":"cloud","input":0.6,"output":2.5,"cache_read":null,"context":262144,"quantization":null},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.6,"output":2.5,"cache_read":0.15,"context":262144,"quantization":"bf16"}]},{"id":"google/gemini-3-flash-preview","slug":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":0.5,"output":3,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.05,"max_output":65536,"speed":null,"modalities":["text","image","file","audio","video"],"features":["json","reasoning","tools"],"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","url":"https://openrouter.ai/google/gemini-3-flash-preview","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google","tag":"google-vertex/global/flex","kind":"cloud","input":0.25,"output":1.5,"cache_read":0.025,"context":1048576,"quantization":null},{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.25,"output":1.5,"cache_read":0.025,"context":1048576,"quantization":null}]},{"id":"deepseek/deepseek-r1","slug":"deepseek-r1","name":"R1","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.7,"output":2.5,"context":64000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":16000,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","url":"https://openrouter.ai/deepseek/deepseek-r1","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.7,"output":2.5,"cache_read":null,"context":64000,"quantization":"fp8"}]},{"id":"openai/gpt-3.5-turbo-0613","slug":"gpt-3-5-turbo-0613","name":"GPT-3.5 Turbo (older v0613)","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1,"output":2,"context":4095,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":3685,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks. Training data up to Sep 2021.","url":"https://openrouter.ai/openai/gpt-3.5-turbo-0613","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":1,"output":2,"cache_read":null,"context":4095,"quantization":null}]},{"id":"x-ai/grok-build-0.1","slug":"grok-build-0-1","name":"Grok Build 0.1","vendor":"x-ai","vendor_name":"xAI","provider":"OpenRouter","provider_id":"openrouter","family":"grok","input":1,"output":2,"context":256000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.2,"max_output":230400,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","url":"https://openrouter.ai/x-ai/grok-build-0.1","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"xAI","tag":"xai","kind":"host","input":1,"output":2,"cache_read":0.2,"context":256000,"quantization":null}]},{"id":"qwen/qwen3.5-397b-a17b","slug":"qwen3-5-397b-a17b","name":"Qwen3.5 397B A17B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.55,"output":3.5,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.225,"max_output":235929,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","url":"https://openrouter.ai/qwen/qwen3.5-397b-a17b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.39,"output":2.34,"cache_read":null,"context":262144,"quantization":null},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":0.45,"output":3,"cache_read":0.22,"context":262144,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":0.5,"output":3.6,"cache_read":0.3,"context":262144,"quantization":"fp8"},{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":0.55,"output":3.5,"cache_read":0.11,"context":131072,"quantization":null},{"provider":"Phala","tag":"phala","kind":"host","input":0.55,"output":3.5,"cache_read":0.225,"context":262144,"quantization":null},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.55,"output":3.5,"cache_read":0.55,"context":262144,"quantization":"fp8"},{"provider":"StreamLake","tag":"streamlake","kind":"host","input":0.6,"output":3.6,"cache_read":0.12,"context":256000,"quantization":null},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.6,"output":3.6,"cache_read":null,"context":262144,"quantization":"fp8"},{"provider":"Novita","tag":"novita","kind":"host","input":0.6,"output":3.6,"cache_read":null,"context":262144,"quantization":null},{"provider":"Venice","tag":"venice","kind":"host","input":0.75,"output":4.5,"cache_read":null,"context":128000,"quantization":null}]},{"id":"qwen/qwen3-coder-plus","slug":"qwen3-coder-plus","name":"Qwen3 Coder Plus","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.65,"output":3.25,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.13,"max_output":65536,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","url":"https://openrouter.ai/qwen/qwen3-coder-plus","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.65,"output":3.25,"cache_read":0.13,"context":1000000,"quantization":null}]},{"id":"qwen/qwen3-vl-235b-a22b-thinking","slug":"qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.4,"output":4,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":32768,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","url":"https://openrouter.ai/qwen/qwen3-vl-235b-a22b-thinking","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.4,"output":4,"cache_read":null,"context":131072,"quantization":null},{"provider":"Novita","tag":"novita/bf16","kind":"host","input":0.98,"output":3.95,"cache_read":null,"context":131072,"quantization":"bf16"}]},{"id":"amazon/nova-pro-v1","slug":"nova-pro-v1","name":"Nova Pro 1.0","vendor":"amazon","vendor_name":"Amazon","provider":"OpenRouter","provider_id":"openrouter","family":"nova","input":0.8,"output":3.2,"context":300000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":5120,"speed":null,"modalities":["text","image"],"features":["tools"],"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","url":"https://openrouter.ai/amazon/nova-pro-v1","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Amazon Bedrock","tag":"amazon-bedrock/eu-west-1","kind":"cloud","input":0.8,"output":3.2,"cache_read":null,"context":300000,"quantization":null}]},{"id":"moonshotai/kimi-k2.7-code","slug":"kimi-k2-7-code","name":"Kimi K2.7 Code","vendor":"moonshotai","vendor_name":"Moonshot","provider":"OpenRouter","provider_id":"openrouter","family":"kimi","input":0.71,"output":3.5,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.15,"max_output":235929,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","url":"https://openrouter.ai/moonshotai/kimi-k2.7-code","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"StreamLake","tag":"streamlake","kind":"host","input":0.7125,"output":3,"cache_read":0.1425,"context":256000,"quantization":null},{"provider":"Inceptron","tag":"inceptron/int4","kind":"host","input":0.7062,"output":3.21,"cache_read":0.18,"context":262144,"quantization":"int4"},{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.68,"output":3.4,"cache_read":0.136,"context":262144,"quantization":"fp4"},{"provider":"CoreWeave","tag":"coreweave/int4","kind":"host","input":0.71,"output":3.5,"cache_read":0.15,"context":262144,"quantization":"int4"},{"provider":"Venice","tag":"venice/int4","kind":"host","input":0.75,"output":3.5,"cache_read":0.16,"context":256000,"quantization":"int4"},{"provider":"ModelRun","tag":"modelrun/fp4","kind":"host","input":0.85,"output":3.75,"cache_read":0.16,"context":262144,"quantization":"fp4"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.8592,"output":3.8,"cache_read":0.1799,"context":262144,"quantization":"fp8"},{"provider":"Novita","tag":"novita/int4","kind":"host","input":0.912,"output":3.84,"cache_read":0.1824,"context":262144,"quantization":"int4"},{"provider":"Fireworks","tag":"fireworks","kind":"host","input":0.95,"output":4,"cache_read":0.19,"context":262144,"quantization":null},{"provider":"BaseTen","tag":"baseten/fp4","kind":"host","input":0.95,"output":4,"cache_read":0.16,"context":262000,"quantization":"fp4"},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.95,"output":4,"cache_read":0.19,"context":262144,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba/fp8","kind":"host","input":0.95,"output":4,"cache_read":0.19,"context":262144,"quantization":"fp8"},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.95,"output":4,"cache_read":0.19,"context":262144,"quantization":null},{"provider":"Moonshot AI","tag":"moonshotai/int4","kind":"host","input":0.95,"output":4,"cache_read":0.19,"context":262144,"quantization":"int4"}]},{"id":"deepseek/deepseek-v4-pro-0813","slug":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":0.9834,"output":2.9502,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.0328,"max_output":384000,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","url":"https://openrouter.ai/deepseek/deepseek-v4-pro-0813","direct_input":1.32,"direct_output":3.96,"direct_cache_read":0.044,"direct_source":"deepseek","routes":[{"provider":"Ionstream","tag":"ionstream","kind":"host","input":0.96,"output":2.88,"cache_read":0.088,"context":1048576,"quantization":null},{"provider":"StreamLake","tag":"streamlake","kind":"host","input":0.9834,"output":2.9502,"cache_read":0.0328,"context":1024000,"quantization":null},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.99,"output":2.97,"cache_read":0.033,"context":1048576,"quantization":"fp8"},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":1.056,"output":3.168,"cache_read":0.0352,"context":1048575,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":1.3,"output":2.6,"cache_read":0.1,"context":1048576,"quantization":"fp8"},{"provider":"NextBit","tag":"nextbit/fp8","kind":"host","input":1.122,"output":3.366,"cache_read":0.037,"context":1048576,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":1.122,"output":3.366,"cache_read":0.1122,"context":1000000,"quantization":null},{"provider":"CoreWeave","tag":"coreweave/fp8","kind":"host","input":1.31,"output":3.96,"cache_read":0.044,"context":1048576,"quantization":"fp8"},{"provider":"Baidu","tag":"baidu/fp8","kind":"host","input":1.32,"output":3.96,"cache_read":0.042,"context":1048576,"quantization":"fp8"},{"provider":"Sail Research","tag":"sail-research/fp4","kind":"host","input":1.32,"output":3.96,"cache_read":0.044,"context":1048576,"quantization":"fp4"},{"provider":"BaseTen","tag":"baseten/fp4","kind":"host","input":1.32,"output":3.96,"cache_read":0.132,"context":1048576,"quantization":"fp4"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":1.32,"output":3.96,"cache_read":0.044,"context":1048576,"quantization":"fp8"},{"provider":"Together","tag":"together","kind":"host","input":1.32,"output":3.96,"cache_read":0.13,"context":1048576,"quantization":null},{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":1.32,"output":3.96,"cache_read":0.044,"context":1048576,"quantization":null},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":1.32,"output":3.96,"cache_read":0.044,"context":1048576,"quantization":"fp8"},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":1.32,"output":3.96,"cache_read":0.044,"context":1048576,"quantization":null},{"provider":"Fireworks","tag":"fireworks","kind":"host","input":1.32,"output":3.96,"cache_read":0.044,"context":1048576,"quantization":null},{"provider":"DeepSeek","tag":"deepseek","kind":"official","input":1.32,"output":3.96,"cache_read":0.044,"context":1048576,"quantization":null},{"provider":"Phala","tag":"phala","kind":"host","input":1.45,"output":4.36,"cache_read":0.15,"context":1048576,"quantization":null},{"provider":"Venice","tag":"venice","kind":"host","input":1.65,"output":4.95,"cache_read":0.165,"context":1000000,"quantization":null}]},{"id":"z-ai/glm-5.1","slug":"glm-5-1","name":"GLM 5.1","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":0.966,"output":3.036,"context":204800,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.1794,"max_output":128000,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","url":"https://openrouter.ai/z-ai/glm-5.1","direct_input":1.4,"direct_output":4.4,"direct_cache_read":0.26,"direct_source":"z-ai","routes":[{"provider":"StreamLake","tag":"streamlake/fp8","kind":"host","input":0.966,"output":3.036,"cache_read":0.1794,"context":200000,"quantization":"fp8"},{"provider":"Chutes","tag":"chutes/fp8","kind":"host","input":0.98,"output":3.08,"cache_read":0.098,"context":202752,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":1.05,"output":3.5,"cache_read":0.205,"context":202752,"quantization":"fp4"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":1.19,"output":3.74,"cache_read":0.6,"context":204800,"quantization":"fp8"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":1.26,"output":3.96,"cache_read":0.234,"context":202752,"quantization":"fp8"},{"provider":"Phala","tag":"phala","kind":"host","input":1.21,"output":4.2,"cache_read":0.6,"context":202752,"quantization":null},{"provider":"Alibaba","tag":"alibaba/fp8","kind":"host","input":1.33,"output":4.18,"cache_read":0.247,"context":202745,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":1.38,"output":4.4,"cache_read":0.26,"context":204800,"quantization":"fp8"},{"provider":"Nebius","tag":"nebius/fp8","kind":"host","input":1.4,"output":4.4,"cache_read":null,"context":202752,"quantization":"fp8"},{"provider":"Baidu","tag":"baidu/fp8","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":202752,"quantization":"fp8"},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":202752,"quantization":"fp8"},{"provider":"Friendli","tag":"friendli","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":202752,"quantization":null},{"provider":"Z.AI","tag":"z-ai/fp8","kind":"official","input":1.4,"output":4.4,"cache_read":0.26,"context":202752,"quantization":"fp8"},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":1.4014,"output":4.4044,"cache_read":0.2603,"context":200000,"quantization":"fp8"}]},{"id":"google/gemini-3.6-flash","slug":"gemini-3-6-flash","name":"Gemini 3.6 Flash","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":0.75,"output":3.75,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.075,"max_output":65536,"speed":null,"modalities":["text","image","video","file","audio"],"features":["json","reasoning","tools"],"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","url":"https://openrouter.ai/google/gemini-3.6-flash","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google","tag":"google-vertex/global/flex","kind":"cloud","input":0.375,"output":1.875,"cache_read":0.0375,"context":1048576,"quantization":null},{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.375,"output":1.875,"cache_read":0.0375,"context":1048576,"quantization":null}]},{"id":"google/gemini-3.7-flash","slug":"gemini-3-7-flash","name":"Gemini 3.7 Flash","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":0.75,"output":3.75,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.075,"max_output":65536,"speed":null,"modalities":["text","image","video","file","audio"],"features":["json","reasoning","tools"],"description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","url":"https://openrouter.ai/google/gemini-3.7-flash","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google","tag":"google-vertex/global/flex","kind":"cloud","input":0.375,"output":1.875,"cache_read":0.0375,"context":1048576,"quantization":null},{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.375,"output":1.875,"cache_read":0.0375,"context":1048576,"quantization":null}]},{"id":"google/gemini-3.8-flash","slug":"gemini-3-8-flash","name":"Gemini 3.8 Flash","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":0.75,"output":3.75,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.075,"max_output":65536,"speed":null,"modalities":["text","image","video","file","audio"],"features":["json","reasoning","tools"],"description":"Gemini 3.8 Flash is Google's most intelligent Flash model with significant gains from 3.7 Flash across software engineering, agentic tasks, and multi-step reasoning.","url":"https://openrouter.ai/google/gemini-3.8-flash","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.375,"output":1.875,"cache_read":0.0375,"context":1048576,"quantization":null},{"provider":"Google","tag":"google-vertex/global/flex","kind":"cloud","input":0.375,"output":1.875,"cache_read":0.0375,"context":1048576,"quantization":null}]},{"id":"qwen/qwen3-max","slug":"qwen3-max","name":"Qwen3 Max","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.78,"output":3.9,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.156,"max_output":65536,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","url":"https://openrouter.ai/qwen/qwen3-max","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.78,"output":3.9,"cache_read":0.156,"context":262144,"quantization":null}]},{"id":"qwen/qwen3-max-thinking","slug":"qwen3-max-thinking","name":"Qwen3 Max Thinking","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":0.78,"output":3.9,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":65536,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","url":"https://openrouter.ai/qwen/qwen3-max-thinking","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":0.78,"output":3.9,"cache_read":null,"context":262144,"quantization":null}]},{"id":"x-ai/grok-4.20","slug":"grok-4-20","name":"Grok 4.20","vendor":"x-ai","vendor_name":"xAI","provider":"OpenRouter","provider_id":"openrouter","family":"grok","input":1.25,"output":2.5,"context":2000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.2,"max_output":1800000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","url":"https://openrouter.ai/x-ai/grok-4.20","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"xAI","tag":"xai/zdr","kind":"host","input":1.25,"output":2.5,"cache_read":0.2,"context":2000000,"quantization":null}]},{"id":"x-ai/grok-4.20-multi-agent","slug":"grok-4-20-multi-agent","name":"Grok 4.20 Multi-Agent","vendor":"x-ai","vendor_name":"xAI","provider":"OpenRouter","provider_id":"openrouter","family":"grok","input":1.25,"output":2.5,"context":2000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.2,"max_output":1800000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning"],"description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","url":"https://openrouter.ai/x-ai/grok-4.20-multi-agent","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"xAI","tag":"xai/zdr","kind":"host","input":1.25,"output":2.5,"cache_read":0.2,"context":2000000,"quantization":null}]},{"id":"x-ai/grok-4.3","slug":"grok-4-3","name":"Grok 4.3","vendor":"x-ai","vendor_name":"xAI","provider":"OpenRouter","provider_id":"openrouter","family":"grok","input":1.25,"output":2.5,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.2,"max_output":900000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","url":"https://openrouter.ai/x-ai/grok-4.3","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"xAI","tag":"xai/zdr","kind":"host","input":1.25,"output":2.5,"cache_read":0.2,"context":1000000,"quantization":null}]},{"id":"openai/gpt-3.5-turbo-instruct","slug":"gpt-3-5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.5,"output":2,"context":4095,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":3685,"speed":null,"modalities":["text"],"features":["json"],"description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","url":"https://openrouter.ai/openai/gpt-3.5-turbo-instruct","direct_input":1.5,"direct_output":2,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":1.5,"output":2,"cache_read":null,"context":4095,"quantization":null}]},{"id":"openai/gpt-5.4-mini","slug":"gpt-5-4-mini","name":"GPT-5.4 Mini","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":0.75,"output":4.5,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.075,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","url":"https://openrouter.ai/openai/gpt-5.4-mini","direct_input":0.375,"direct_output":2.25,"direct_cache_read":0.0375,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":0.375,"output":2.25,"cache_read":0.0375,"context":400000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":0.75,"output":4.5,"cache_read":0.075,"context":400000,"quantization":null}]},{"id":"moonshotai/kimi-k2.6","slug":"kimi-k2-6","name":"Kimi K2.6","vendor":"moonshotai","vendor_name":"Moonshot","provider":"OpenRouter","provider_id":"openrouter","family":"kimi","input":0.95,"output":4,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.16,"max_output":235929,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","url":"https://openrouter.ai/moonshotai/kimi-k2.6","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":0.57,"output":2.4,"cache_read":0.114,"context":262144,"quantization":null},{"provider":"Decart","tag":"decart/fp4","kind":"host","input":0.5865,"output":2.4696,"cache_read":0.0988,"context":262144,"quantization":"fp4"},{"provider":"StreamLake","tag":"streamlake/fp8","kind":"host","input":0.5985,"output":2.52,"cache_read":0.1008,"context":256000,"quantization":"fp8"},{"provider":"Inceptron","tag":"inceptron/int4","kind":"host","input":0.602,"output":2.87,"cache_read":0.1144,"context":262144,"quantization":"int4"},{"provider":"Chutes","tag":"chutes/int4","kind":"host","input":0.58,"output":3.4,"cache_read":0.058,"context":262144,"quantization":"int4"},{"provider":"CoreWeave","tag":"coreweave/fp4","kind":"host","input":0.65,"output":3.41,"cache_read":0.15,"context":262144,"quantization":"fp4"},{"provider":"Crusoe","tag":"crusoe/bf16","kind":"host","input":0.7,"output":3.5,"cache_read":0.35,"context":262144,"quantization":"bf16"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":0.77,"output":3.4,"cache_read":0.14,"context":262144,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.75,"output":3.5,"cache_read":0.15,"context":262144,"quantization":"fp4"},{"provider":"Venice","tag":"venice/int4","kind":"host","input":0.75,"output":3.5,"cache_read":0.16,"context":256000,"quantization":"int4"},{"provider":"Parasail","tag":"parasail/int4","kind":"host","input":0.75,"output":3.5,"cache_read":0.16,"context":262144,"quantization":"int4"},{"provider":"Novita","tag":"novita","kind":"host","input":0.8,"output":3.4,"cache_read":0.16,"context":262144,"quantization":null},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.855,"output":3.6,"cache_read":0.144,"context":262144,"quantization":"fp8"},{"provider":"Baidu","tag":"baidu/fp4","kind":"host","input":0.95,"output":4,"cache_read":0.16,"context":262144,"quantization":"fp4"},{"provider":"AtlasCloud","tag":"atlas-cloud/int4","kind":"host","input":0.95,"output":4,"cache_read":0.16,"context":262144,"quantization":"int4"},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":0.95,"output":4,"cache_read":0.16,"context":262144,"quantization":null},{"provider":"Moonshot AI","tag":"moonshotai/int4","kind":"host","input":0.95,"output":4,"cache_read":0.16,"context":262144,"quantization":"int4"},{"provider":"BaseTen","tag":"baseten/fp4","kind":"host","input":0.95,"output":4,"cache_read":0.16,"context":262000,"quantization":"fp4"},{"provider":"Fireworks","tag":"fireworks","kind":"host","input":0.95,"output":4,"cache_read":0.16,"context":262144,"quantization":null},{"provider":"Phala","tag":"phala","kind":"host","input":1.09,"output":4.6,"cache_read":0.37,"context":262144,"quantization":null}]},{"id":"z-ai/glm-5-turbo","slug":"glm-5-turbo","name":"GLM 5 Turbo","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":1.2,"output":4,"context":202752,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.24,"max_output":131072,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","url":"https://openrouter.ai/z-ai/glm-5-turbo","direct_input":1.2,"direct_output":4,"direct_cache_read":0.24,"direct_source":"z-ai","routes":[{"provider":"Z.AI","tag":"z-ai","kind":"official","input":1.2,"output":4,"cache_read":0.24,"context":202752,"quantization":null}]},{"id":"z-ai/glm-5v-turbo","slug":"glm-5v-turbo","name":"GLM 5V Turbo","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":1.2,"output":4,"context":202752,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.24,"max_output":131072,"speed":null,"modalities":["image","text","video"],"features":["json","reasoning","tools"],"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","url":"https://openrouter.ai/z-ai/glm-5v-turbo","direct_input":1.2,"direct_output":4,"direct_cache_read":0.24,"direct_source":"z-ai","routes":[{"provider":"Z.AI","tag":"z-ai/fp8","kind":"official","input":1.2,"output":4,"cache_read":0.24,"context":202752,"quantization":"fp8"}]},{"id":"openai/o3-mini","slug":"o3-mini","name":"o3 Mini","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.1,"output":4.4,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.55,"max_output":100000,"speed":null,"modalities":["text","file"],"features":["json","reasoning","tools"],"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","url":"https://openrouter.ai/openai/o3-mini","direct_input":1.1,"direct_output":4.4,"direct_cache_read":0.55,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":1.1,"output":4.4,"cache_read":0.55,"context":200000,"quantization":null}]},{"id":"openai/o3-mini-high","slug":"o3-mini-high","name":"o3 Mini High","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.1,"output":4.4,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.55,"max_output":100000,"speed":null,"modalities":["text","file"],"features":["json","reasoning","tools"],"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","url":"https://openrouter.ai/openai/o3-mini-high","direct_input":1.1,"direct_output":4.4,"direct_cache_read":0.55,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":1.1,"output":4.4,"cache_read":0.55,"context":200000,"quantization":null}]},{"id":"openai/o4-mini","slug":"o4-mini","name":"o4 Mini","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.1,"output":4.4,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.275,"max_output":100000,"speed":null,"modalities":["image","text","file"],"features":["json","reasoning","tools"],"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","url":"https://openrouter.ai/openai/o4-mini","direct_input":1.1,"direct_output":4.4,"direct_cache_read":0.275,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":1.1,"output":4.4,"cache_read":0.275,"context":200000,"quantization":null}]},{"id":"openai/o4-mini-high","slug":"o4-mini-high","name":"o4 Mini High","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.1,"output":4.4,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.275,"max_output":100000,"speed":null,"modalities":["image","text","file"],"features":["json","reasoning","tools"],"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","url":"https://openrouter.ai/openai/o4-mini-high","direct_input":1.1,"direct_output":4.4,"direct_cache_read":0.275,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":1.1,"output":4.4,"cache_read":0.275,"context":200000,"quantization":null}]},{"id":"anthropic/claude-haiku-4.5","slug":"claude-haiku-4-5","name":"Claude Haiku 4.5","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":1,"output":5,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.1,"max_output":64000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","url":"https://openrouter.ai/anthropic/claude-haiku-4.5","direct_input":1,"direct_output":5,"direct_cache_read":0.1,"direct_source":"anthropic","routes":[{"provider":"Azure","tag":"azure/global","kind":"cloud","input":1,"output":5,"cache_read":0.1,"context":200000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/global","kind":"cloud","input":1,"output":5,"cache_read":0.1,"context":200000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":1,"output":5,"cache_read":0.1,"context":200000,"quantization":null},{"provider":"Anthropic","tag":"anthropic","kind":"official","input":1,"output":5,"cache_read":0.1,"context":200000,"quantization":null}]},{"id":"deepseek/deepseek-v4-pro","slug":"deepseek-v4-pro","name":"DeepSeek V4 Pro 0423","vendor":"deepseek","vendor_name":"DeepSeek","provider":"OpenRouter","provider_id":"openrouter","family":"deepseek","input":1.6,"output":3.2,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.135,"max_output":393216,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","url":"https://openrouter.ai/deepseek/deepseek-v4-pro","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"StreamLake","tag":"streamlake/fp8","kind":"host","input":0.9553,"output":1.9105,"cache_read":0.0796,"context":1024000,"quantization":"fp8"},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":0.957,"output":1.914,"cache_read":0.0798,"context":1048576,"quantization":"fp8"},{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":1.044,"output":2.088,"cache_read":0.2088,"context":1048576,"quantization":null},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":1.15,"output":2.55,"cache_read":0.2,"context":1048576,"quantization":null},{"provider":"DeepInfra","tag":"deepinfra/fp8","kind":"host","input":1.3,"output":2.6,"cache_read":0.1,"context":1048576,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba/fp8","kind":"host","input":1.416,"output":2.832,"cache_read":0.118,"context":1000000,"quantization":"fp8"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":1.5016,"output":3.135,"cache_read":0.135,"context":1048576,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":1.6,"output":3.2,"cache_read":0.135,"context":1048576,"quantization":"fp8"},{"provider":"Venice","tag":"venice","kind":"host","input":1.65,"output":3.301,"cache_read":0.33,"context":1000000,"quantization":null},{"provider":"AtlasCloud","tag":"atlas-cloud/fp4","kind":"host","input":1.68,"output":3.38,"cache_read":0.13,"context":1048576,"quantization":"fp4"},{"provider":"Baidu","tag":"baidu/fp8","kind":"host","input":1.69,"output":3.38,"cache_read":0.14,"context":1048576,"quantization":"fp8"},{"provider":"BaseTen","tag":"baseten/fp4","kind":"host","input":1.74,"output":3.48,"cache_read":0.145,"context":1048576,"quantization":"fp4"},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":1.74,"output":3.48,"cache_read":0.1,"context":1048576,"quantization":"fp8"},{"provider":"Azure","tag":"azure/us","kind":"cloud","input":1.91,"output":3.83,"cache_read":0.16,"context":1048576,"quantization":null},{"provider":"NextBit","tag":"nextbit/fp8","kind":"host","input":2.175,"output":4.35,"cache_read":0.18,"context":1048576,"quantization":"fp8"}]},{"id":"z-ai/glm-5.2","slug":"glm-5-2","name":"GLM 5.2","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":1.4,"output":4.4,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.14,"max_output":128000,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","url":"https://openrouter.ai/z-ai/glm-5.2","direct_input":1.4,"direct_output":4.4,"direct_cache_read":0.26,"direct_source":"z-ai","routes":[{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.4875,"output":1.56,"cache_read":0.091,"context":1048576,"quantization":"fp4"},{"provider":"StreamLake","tag":"streamlake/fp8","kind":"host","input":0.56,"output":1.76,"cache_read":0.104,"context":1024000,"quantization":"fp8"},{"provider":"Ambient","tag":"ambient/fp8","kind":"host","input":0.6,"output":2,"cache_read":0.15,"context":202752,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":0.6832,"output":2.1472,"cache_read":0.1269,"context":1048576,"quantization":"fp8"},{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":0.7,"output":2.2,"cache_read":0.105,"context":262144,"quantization":null},{"provider":"CoreWeave","tag":"coreweave/fp4","kind":"host","input":0.76,"output":2.42,"cache_read":0.14,"context":1048576,"quantization":"fp4"},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":0.938,"output":2.948,"cache_read":0.1742,"context":1048576,"quantization":"fp8"},{"provider":"Alibaba","tag":"alibaba/fp8","kind":"host","input":0.966,"output":3.036,"cache_read":0.1932,"context":1000000,"quantization":"fp8"},{"provider":"Inceptron","tag":"inceptron/fp4","kind":"host","input":1.0998,"output":2.9905,"cache_read":0.18,"context":1048576,"quantization":"fp4"},{"provider":"Phala","tag":"phala/fp8","kind":"host","input":1.26,"output":3,"cache_read":0.22,"context":1048576,"quantization":"fp8"},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":1.19,"output":3.74,"cache_read":0.221,"context":1048576,"quantization":"fp8"},{"provider":"Mistral","tag":"mistral/zdr","kind":"host","input":1.4,"output":4.4,"cache_read":0.14,"context":1048576,"quantization":null},{"provider":"BaseTen","tag":"baseten/fp8","kind":"host","input":1.4,"output":4.4,"cache_read":0.14,"context":1048576,"quantization":"fp8"},{"provider":"Together","tag":"together","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":512000,"quantization":null},{"provider":"Baidu","tag":"baidu/fp8","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":"fp8"},{"provider":"Fireworks","tag":"fireworks","kind":"host","input":1.4,"output":4.4,"cache_read":0.14,"context":1048576,"quantization":null},{"provider":"Venice","tag":"venice/fp8","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1000000,"quantization":"fp8"},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":"fp8"},{"provider":"Parasail","tag":"parasail/fp4","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":262144,"quantization":"fp4"},{"provider":"Friendli","tag":"friendli","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":null},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":262144,"quantization":null},{"provider":"Z.AI","tag":"z-ai/fp8","kind":"official","input":1.4,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":"fp8"},{"provider":"Decart","tag":"decart/fast","kind":"host","input":2.1,"output":6.6,"cache_read":0.21,"context":1048576,"quantization":"fp4"}]},{"id":"z-ai/glm-5.3","slug":"glm-5-3","name":"GLM 5.3","vendor":"z-ai","vendor_name":"Zhipu","provider":"OpenRouter","provider_id":"openrouter","family":"glm","input":1.4,"output":4.4,"context":1310720,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.26,"max_output":943717,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","url":"https://openrouter.ai/z-ai/glm-5.3","direct_input":1.4,"direct_output":4.4,"direct_cache_read":0.26,"direct_source":"z-ai","routes":[{"provider":"Reka","tag":"reka/fp8","kind":"host","input":0.8775,"output":2.97,"cache_read":0.1755,"context":262144,"quantization":"fp8"},{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":0.9,"output":3,"cache_read":0.15,"context":1048576,"quantization":"fp4"},{"provider":"Morph","tag":"morph/fp8","kind":"host","input":0.95,"output":3.2395,"cache_read":0.19,"context":1048576,"quantization":"fp8"},{"provider":"Novita","tag":"novita/fp8","kind":"host","input":1.036,"output":3.256,"cache_read":0.1924,"context":1048576,"quantization":"fp8"},{"provider":"Phala","tag":"phala","kind":"host","input":1.12,"output":3.52,"cache_read":0.208,"context":1048576,"quantization":null},{"provider":"GMICloud","tag":"gmicloud/fp8","kind":"host","input":1.12,"output":3.52,"cache_read":0.208,"context":1048576,"quantization":"fp8"},{"provider":"Wafer","tag":"wafer","kind":"host","input":0.95,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":null},{"provider":"Decart","tag":"decart/fp4","kind":"host","input":1.19,"output":3.74,"cache_read":0.1955,"context":1048576,"quantization":"fp4"},{"provider":"Inceptron","tag":"inceptron/fp4","kind":"host","input":1.059,"output":4.136,"cache_read":0.1882,"context":1048576,"quantization":"fp4"},{"provider":"Sail Research","tag":"sail-research/fp8","kind":"host","input":1.2572,"output":3.9512,"cache_read":0.2335,"context":1048576,"quantization":"fp8"},{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":1.26,"output":3.96,"cache_read":0.234,"context":1048576,"quantization":null},{"provider":"Friendli","tag":"friendli","kind":"host","input":1.26,"output":3.96,"cache_read":0.234,"context":1048576,"quantization":null},{"provider":"Io Net","tag":"io-net/fp8","kind":"host","input":1.1875,"output":4.56,"cache_read":0.2755,"context":262124,"quantization":"fp8"},{"provider":"AkashML","tag":"akashml/fp8","kind":"host","input":1.3,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":"fp8"},{"provider":"Makora","tag":"makora/fp4","kind":"host","input":1.35,"output":4.4,"cache_read":0.23,"context":980000,"quantization":"fp4"},{"provider":"Baidu","tag":"baidu/fp8","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":"fp8"},{"provider":"Crusoe","tag":"crusoe/fp4","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":"fp4"},{"provider":"Venice","tag":"venice","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1000000,"quantization":null},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":"fp8"},{"provider":"Together","tag":"together","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1048575,"quantization":null},{"provider":"Parasail","tag":"parasail/fp8","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":"fp8"},{"provider":"Modal","tag":"modal","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":null},{"provider":"BaseTen","tag":"baseten/fp4","kind":"host","input":1.4,"output":4.4,"cache_read":0.14,"context":1048576,"quantization":"fp4"},{"provider":"Fireworks","tag":"fireworks","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":null},{"provider":"Cloudflare","tag":"cloudflare","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":1310720,"quantization":null},{"provider":"AtlasCloud","tag":"atlas-cloud/fp8","kind":"host","input":1.4,"output":4.4,"cache_read":0.26,"context":262144,"quantization":"fp8"},{"provider":"Z.AI","tag":"z-ai/fp8","kind":"official","input":1.4,"output":4.4,"cache_read":0.26,"context":1048576,"quantization":"fp8"}]},{"id":"qwen/qwen3.7-max","slug":"qwen3-7-max","name":"Qwen3.7 Max","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":1.475,"output":4.425,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.295,"max_output":131072,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","url":"https://openrouter.ai/qwen/qwen3.7-max","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":1.475,"output":4.425,"cache_read":0.295,"context":1000000,"quantization":null}]},{"id":"qwen/qwen3.6-max-preview","slug":"qwen3-6-max-preview","name":"Qwen3.6 Max Preview","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":1.027,"output":6.162,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":65536,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","url":"https://openrouter.ai/qwen/qwen3.6-max-preview","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":1.027,"output":6.162,"cache_read":null,"context":262144,"quantization":null}]},{"id":"mistralai/mistral-large-2407","slug":"mistral-large-2407","name":"Mistral Large 2407","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":2,"output":6,"context":131072,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.2,"max_output":104857,"speed":null,"modalities":["text","file"],"features":["json","tools"],"description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","url":"https://openrouter.ai/mistralai/mistral-large-2407","direct_input":2,"direct_output":6,"direct_cache_read":0.2,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/zdr","kind":"official","input":2,"output":6,"cache_read":0.2,"context":131072,"quantization":null}]},{"id":"mistralai/mistral-large","slug":"mistral-large","name":"Mistral Large","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":2,"output":6,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.2,"max_output":102400,"speed":null,"modalities":["text","file"],"features":["json","tools"],"description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","url":"https://openrouter.ai/mistralai/mistral-large","direct_input":2,"direct_output":6,"direct_cache_read":0.2,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral","kind":"official","input":2,"output":6,"cache_read":0.2,"context":128000,"quantization":null}]},{"id":"mistralai/mistral-medium-3-5","slug":"mistral-medium-3-5","name":"Mistral Medium 3.5","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":1.5,"output":7.5,"context":262144,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":209715,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","url":"https://openrouter.ai/mistralai/mistral-medium-3-5","direct_input":1.5,"direct_output":7.5,"direct_cache_read":null,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/zdr","kind":"official","input":1.5,"output":7.5,"cache_read":null,"context":262144,"quantization":null}]},{"id":"mistralai/mixtral-8x22b-instruct","slug":"mixtral-8x22b-instruct","name":"Mixtral 8x22B Instruct","vendor":"mistralai","vendor_name":"Mistral","provider":"OpenRouter","provider_id":"openrouter","family":"mistral","input":2,"output":6,"context":65536,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.2,"max_output":52428,"speed":null,"modalities":["text","file"],"features":["json","tools"],"description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","url":"https://openrouter.ai/mistralai/mixtral-8x22b-instruct","direct_input":2,"direct_output":6,"direct_cache_read":0.2,"direct_source":"mistralai","routes":[{"provider":"Mistral","tag":"mistral/zdr","kind":"official","input":2,"output":6,"cache_read":0.2,"context":65536,"quantization":null}]},{"id":"qwen/qwen3.8-2.4t-a95b","slug":"qwen3-8-2-4t-a95b","name":"Qwen3.8 2.4T A95B","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":2,"output":6,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.25,"max_output":131072,"speed":null,"modalities":["text"],"features":["json","reasoning","tools"],"description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","url":"https://openrouter.ai/qwen/qwen3.8-2.4t-a95b","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Novita","tag":"novita","kind":"host","input":2,"output":6,"cache_read":0.25,"context":1000000,"quantization":null},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":2,"output":6,"cache_read":0.25,"context":1000000,"quantization":null},{"provider":"SiliconFlow","tag":"siliconflow/fp8","kind":"host","input":2,"output":6,"cache_read":0.25,"context":1048576,"quantization":"fp8"},{"provider":"Venice","tag":"venice","kind":"host","input":2,"output":6,"cache_read":0.25,"context":262144,"quantization":null},{"provider":"Modal","tag":"modal","kind":"host","input":2,"output":6,"cache_read":0.25,"context":1000000,"quantization":null},{"provider":"DeepInfra","tag":"deepinfra/fp4","kind":"host","input":2,"output":6,"cache_read":0.2,"context":262144,"quantization":"fp4"},{"provider":"Together","tag":"together","kind":"host","input":2,"output":6,"cache_read":0.25,"context":1010000,"quantization":null}]},{"id":"qwen/qwen3.8-max-0902","slug":"qwen3-8-max-0902","name":"Qwen3.8 Max (0902)","vendor":"qwen","vendor_name":"Alibaba","provider":"OpenRouter","provider_id":"openrouter","family":"qwen","input":2,"output":6,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.25,"max_output":131072,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"Qwen3.8 Max 0902 is an updated snapshot of Qwen3.8 Max from Alibaba's Qwen team. It is a 2.4-trillion-parameter mixture-of-experts model that accepts text, image, and video input and returns text,...","url":"https://openrouter.ai/qwen/qwen3.8-max-0902","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Alibaba","tag":"alibaba","kind":"host","input":2,"output":6,"cache_read":0.25,"context":1000000,"quantization":null}]},{"id":"x-ai/grok-4.5","slug":"grok-4-5","name":"Grok 4.5","vendor":"x-ai","vendor_name":"xAI","provider":"OpenRouter","provider_id":"openrouter","family":"grok","input":2,"output":6,"context":500000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.3,"max_output":450000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","url":"https://openrouter.ai/x-ai/grok-4.5","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"xAI","tag":"xai/zdr","kind":"host","input":2,"output":6,"cache_read":0.3,"context":500000,"quantization":null}]},{"id":"x-ai/grok-4.6","slug":"grok-4-6","name":"Grok 4.6","vendor":"x-ai","vendor_name":"xAI","provider":"OpenRouter","provider_id":"openrouter","family":"grok","input":2,"output":6,"context":500000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.5,"max_output":450000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","url":"https://openrouter.ai/x-ai/grok-4.6","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"xAI","tag":"xai/zdr","kind":"host","input":2,"output":6,"cache_read":0.5,"context":500000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/us-west-2","kind":"cloud","input":2.2,"output":6.6,"cache_read":0.55,"context":500000,"quantization":null}]},{"id":"openai/gpt-3.5-turbo-16k","slug":"gpt-3-5-turbo-16k","name":"GPT-3.5 Turbo 16k","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":3,"output":4,"context":16385,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":4096,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","url":"https://openrouter.ai/openai/gpt-3.5-turbo-16k","direct_input":3,"direct_output":4,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":3,"output":4,"cache_read":null,"context":16385,"quantization":null},{"provider":"OpenAI","tag":"openai","kind":"official","input":3,"output":4,"cache_read":null,"context":16385,"quantization":null}]},{"id":"google/gemini-3.5-flash","slug":"gemini-3-5-flash","name":"Gemini 3.5 Flash","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":1.5,"output":9,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.15,"max_output":65536,"speed":null,"modalities":["text","image","video","file","audio"],"features":["json","reasoning","tools"],"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","url":"https://openrouter.ai/google/gemini-3.5-flash","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google","tag":"google-vertex/global/flex","kind":"cloud","input":0.75,"output":4.5,"cache_read":0.075,"context":1048576,"quantization":null},{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.75,"output":4.5,"cache_read":0.075,"context":1048576,"quantization":null}]},{"id":"google/gemini-2.5-pro","slug":"gemini-2-5-pro","name":"Gemini 2.5 Pro","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":1.25,"output":10,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.125,"max_output":65536,"speed":null,"modalities":["text","image","file","audio","video"],"features":["json","reasoning","tools"],"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","url":"https://openrouter.ai/google/gemini-2.5-pro","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.625,"output":5,"cache_read":0.0625,"context":1048576,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":1.25,"output":10,"cache_read":0.125,"context":1048576,"quantization":null}]},{"id":"google/gemini-2.5-pro-preview","slug":"gemini-2-5-pro-preview","name":"Gemini 2.5 Pro Preview 06-05","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":1.25,"output":10,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":0.125,"max_output":65536,"speed":null,"modalities":["file","image","text","audio"],"features":["json","reasoning","tools"],"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","url":"https://openrouter.ai/google/gemini-2.5-pro-preview","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":0.625,"output":5,"cache_read":0.0625,"context":1048576,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":1.25,"output":10,"cache_read":0.125,"context":1048576,"quantization":null}]},{"id":"openai/gpt-5","slug":"gpt-5","name":"GPT-5","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.25,"output":10,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.125,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","url":"https://openrouter.ai/openai/gpt-5","direct_input":1.25,"direct_output":10,"direct_cache_read":0.125,"direct_source":"openai","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":1.25,"output":10,"cache_read":0.125,"context":400000,"quantization":null},{"provider":"OpenAI","tag":"openai/default","kind":"official","input":1.25,"output":10,"cache_read":0.125,"context":400000,"quantization":null}]},{"id":"openai/gpt-5.1","slug":"gpt-5-1","name":"GPT-5.1","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.25,"output":10,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.125,"max_output":128000,"speed":null,"modalities":["image","text","file"],"features":["json","reasoning","tools"],"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","url":"https://openrouter.ai/openai/gpt-5.1","direct_input":0.625,"direct_output":5,"direct_cache_read":0.0625,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":0.625,"output":5,"cache_read":0.0625,"context":400000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":1.25,"output":10,"cache_read":0.13,"context":400000,"quantization":null}]},{"id":"openai/gpt-5.1-codex","slug":"gpt-5-1-codex","name":"GPT-5.1-Codex","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.25,"output":10,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.13,"max_output":128000,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","url":"https://openrouter.ai/openai/gpt-5.1-codex","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":1.25,"output":10,"cache_read":0.13,"context":400000,"quantization":null}]},{"id":"openai/gpt-5.1-codex-max","slug":"gpt-5-1-codex-max","name":"GPT-5.1-Codex-Max","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.25,"output":10,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.125,"max_output":128000,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","url":"https://openrouter.ai/openai/gpt-5.1-codex-max","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":1.25,"output":10,"cache_read":0.125,"context":400000,"quantization":null}]},{"id":"openai/gpt-4.1","slug":"gpt-4-1","name":"GPT-4.1","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":2,"output":8,"context":1047576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.5,"max_output":32768,"speed":null,"modalities":["image","text","file"],"features":["json","tools"],"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","url":"https://openrouter.ai/openai/gpt-4.1","direct_input":2,"direct_output":8,"direct_cache_read":0.5,"direct_source":"openai","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":2,"output":8,"cache_read":0.5,"context":1047576,"quantization":null},{"provider":"OpenAI","tag":"openai","kind":"official","input":2,"output":8,"cache_read":0.5,"context":1047576,"quantization":null}]},{"id":"openai/o3","slug":"o3","name":"o3","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":2,"output":8,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.5,"max_output":100000,"speed":null,"modalities":["image","text","file"],"features":["json","reasoning","tools"],"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","url":"https://openrouter.ai/openai/o3","direct_input":2,"direct_output":8,"direct_cache_read":0.5,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":2,"output":8,"cache_read":0.5,"context":200000,"quantization":null}]},{"id":"anthropic/claude-sonnet-5","slug":"claude-sonnet-5","name":"Claude Sonnet 5","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":2,"output":10,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.2,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","url":"https://openrouter.ai/anthropic/claude-sonnet-5","direct_input":2,"direct_output":10,"direct_cache_read":0.2,"direct_source":"anthropic","routes":[{"provider":"Claude Platform on AWS","tag":"claude-on-aws","kind":"cloud","input":2,"output":10,"cache_read":0.2,"context":1000000,"quantization":null},{"provider":"Azure","tag":"azure/us","kind":"cloud","input":2,"output":10,"cache_read":0.2,"context":1000000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":2,"output":10,"cache_read":0.2,"context":1000000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/global","kind":"cloud","input":2,"output":10,"cache_read":0.2,"context":1000000,"quantization":null},{"provider":"Anthropic","tag":"anthropic","kind":"official","input":2,"output":10,"cache_read":0.2,"context":1000000,"quantization":null}]},{"id":"openai/gpt-5.6-sol","slug":"gpt-5-6-sol","name":"GPT-5.6 Sol","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":2,"output":10,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.2,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","url":"https://openrouter.ai/openai/gpt-5.6-sol","direct_input":1,"direct_output":5,"direct_cache_read":0.1,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":1,"output":5,"cache_read":0.1,"context":1050000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/us-east-1","kind":"cloud","input":4.4,"output":22,"cache_read":0.44,"context":1050000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":5,"output":30,"cache_read":0.5,"context":1050000,"quantization":null}]},{"id":"openai/gpt-5.6-sol-pro","slug":"gpt-5-6-sol-pro","name":"GPT-5.6 Sol Pro","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":2,"output":10,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.2,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","url":"https://openrouter.ai/openai/gpt-5.6-sol-pro","direct_input":1,"direct_output":5,"direct_cache_read":0.1,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":1,"output":5,"cache_read":0.1,"context":1050000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":5,"output":30,"cache_read":0.5,"context":1050000,"quantization":null}]},{"id":"cohere/command-a","slug":"command-a","name":"Command A","vendor":"cohere","vendor_name":"Cohere","provider":"OpenRouter","provider_id":"openrouter","family":"command","input":2.5,"output":10,"context":256000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":8192,"speed":null,"modalities":["text"],"features":["json"],"description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","url":"https://openrouter.ai/cohere/command-a","direct_input":2.5,"direct_output":10,"direct_cache_read":null,"direct_source":"cohere","routes":[{"provider":"Cohere","tag":"cohere","kind":"official","input":2.5,"output":10,"cache_read":null,"context":256000,"quantization":null}]},{"id":"cohere/command-r-plus-08-2024","slug":"command-r-plus-08-2024","name":"Command R+ (08-2024)","vendor":"cohere","vendor_name":"Cohere","provider":"OpenRouter","provider_id":"openrouter","family":"command","input":2.5,"output":10,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":4000,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","url":"https://openrouter.ai/cohere/command-r-plus-08-2024","direct_input":2.5,"direct_output":10,"direct_cache_read":null,"direct_source":"cohere","routes":[{"provider":"Cohere","tag":"cohere","kind":"official","input":2.5,"output":10,"cache_read":null,"context":128000,"quantization":null}]},{"id":"openai/gpt-4o-2024-11-20","slug":"gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":2.5,"output":10,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":1.25,"max_output":16384,"speed":null,"modalities":["text","image","file"],"features":["json","tools"],"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","url":"https://openrouter.ai/openai/gpt-4o-2024-11-20","direct_input":2.5,"direct_output":10,"direct_cache_read":1.25,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":2.5,"output":10,"cache_read":1.25,"context":128000,"quantization":null}]},{"id":"openai/gpt-4o-2024-08-06","slug":"gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":2.5,"output":10,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":1.25,"max_output":16384,"speed":null,"modalities":["text","image","file"],"features":["json","tools"],"description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","url":"https://openrouter.ai/openai/gpt-4o-2024-08-06","direct_input":2.5,"direct_output":10,"direct_cache_read":1.25,"direct_source":"openai","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":2.5,"output":10,"cache_read":1.25,"context":128000,"quantization":null},{"provider":"OpenAI","tag":"openai","kind":"official","input":2.5,"output":10,"cache_read":1.25,"context":128000,"quantization":null}]},{"id":"openai/gpt-4o","slug":"gpt-4o","name":"GPT-4o","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":2.5,"output":10,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":1.25,"max_output":16384,"speed":null,"modalities":["text","image","file"],"features":["json","tools"],"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","url":"https://openrouter.ai/openai/gpt-4o","direct_input":2.5,"direct_output":10,"direct_cache_read":1.25,"direct_source":"openai","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":2.5,"output":10,"cache_read":null,"context":128000,"quantization":null},{"provider":"OpenAI","tag":"openai","kind":"official","input":2.5,"output":10,"cache_read":1.25,"context":128000,"quantization":null}]},{"id":"google/gemini-3.1-pro-preview","slug":"gemini-3-1-pro-preview","name":"Gemini 3.1 Pro Preview","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":2,"output":12,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.2,"max_output":65536,"speed":null,"modalities":["audio","file","image","text","video"],"features":["json","reasoning","tools"],"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","url":"https://openrouter.ai/google/gemini-3.1-pro-preview","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google","tag":"google-vertex/global/flex","kind":"cloud","input":1,"output":6,"cache_read":0.1,"context":1048576,"quantization":null},{"provider":"Google AI Studio","tag":"google-ai-studio/flex","kind":"host","input":1,"output":6,"cache_read":0.1,"context":1048576,"quantization":null}]},{"id":"google/gemini-3.1-pro-preview-customtools","slug":"gemini-3-1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","vendor":"google","vendor_name":"Google","provider":"OpenRouter","provider_id":"openrouter","family":"gemini","input":2,"output":12,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.2,"max_output":65536,"speed":null,"modalities":["text","audio","image","video","file"],"features":["json","reasoning","tools"],"description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","url":"https://openrouter.ai/google/gemini-3.1-pro-preview-customtools","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google AI Studio","tag":"google-ai-studio","kind":"host","input":2,"output":12,"cache_read":0.2,"context":1048576,"quantization":null}]},{"id":"openai/gpt-5.6-terra","slug":"gpt-5-6-terra","name":"GPT-5.6 Terra","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":2,"output":12,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.2,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","url":"https://openrouter.ai/openai/gpt-5.6-terra","direct_input":1,"direct_output":6,"direct_cache_read":0.1,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":1,"output":6,"cache_read":0.1,"context":1050000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":2,"output":12,"cache_read":0.2,"context":1050000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/us-east-1","kind":"cloud","input":2.2,"output":13.2,"cache_read":0.22,"context":1050000,"quantization":null}]},{"id":"openai/gpt-5.6-terra-pro","slug":"gpt-5-6-terra-pro","name":"GPT-5.6 Terra Pro","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":2,"output":12,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.2,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","url":"https://openrouter.ai/openai/gpt-5.6-terra-pro","direct_input":1,"direct_output":6,"direct_cache_read":0.1,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":1,"output":6,"cache_read":0.1,"context":1050000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":2,"output":12,"cache_read":0.2,"context":1050000,"quantization":null}]},{"id":"openai/gpt-5.2","slug":"gpt-5-2","name":"GPT-5.2","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.75,"output":14,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.175,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","url":"https://openrouter.ai/openai/gpt-5.2","direct_input":0.875,"direct_output":7,"direct_cache_read":0.0875,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":0.875,"output":7,"cache_read":0.0875,"context":400000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":1.75,"output":14,"cache_read":0.175,"context":400000,"quantization":null}]},{"id":"openai/gpt-5.2-chat","slug":"gpt-5-2-chat","name":"GPT-5.2 Chat","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.75,"output":14,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.175,"max_output":32000,"speed":null,"modalities":["file","image","text"],"features":["json","tools"],"description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","url":"https://openrouter.ai/openai/gpt-5.2-chat","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":1.75,"output":14,"cache_read":0.175,"context":128000,"quantization":null}]},{"id":"openai/gpt-5.2-codex","slug":"gpt-5-2-codex","name":"GPT-5.2-Codex","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.75,"output":14,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.175,"max_output":128000,"speed":null,"modalities":["text","image"],"features":["json","reasoning","tools"],"description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","url":"https://openrouter.ai/openai/gpt-5.2-codex","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":1.75,"output":14,"cache_read":0.175,"context":400000,"quantization":null}]},{"id":"openai/gpt-5.3-codex","slug":"gpt-5-3-codex","name":"GPT-5.3-Codex","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":1.75,"output":14,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.175,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","url":"https://openrouter.ai/openai/gpt-5.3-codex","direct_input":1.75,"direct_output":14,"direct_cache_read":0.175,"direct_source":"openai","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":1.75,"output":14,"cache_read":0.175,"context":400000,"quantization":null},{"provider":"OpenAI","tag":"openai","kind":"official","input":1.75,"output":14,"cache_read":0.175,"context":400000,"quantization":null}]},{"id":"amazon/nova-premier-v1","slug":"nova-premier-v1","name":"Nova Premier 1.0","vendor":"amazon","vendor_name":"Amazon","provider":"OpenRouter","provider_id":"openrouter","family":"nova","input":2.5,"output":12.5,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.625,"max_output":32000,"speed":null,"modalities":["text","image"],"features":["tools"],"description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","url":"https://openrouter.ai/amazon/nova-premier-v1","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":2.5,"output":12.5,"cache_read":0.625,"context":1000000,"quantization":null}]},{"id":"moonshotai/kimi-k3","slug":"kimi-k3","name":"Kimi K3","vendor":"moonshotai","vendor_name":"Moonshot","provider":"OpenRouter","provider_id":"openrouter","family":"kimi","input":2.6481,"output":13.2827,"context":1048576,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.3026,"max_output":943718,"speed":null,"modalities":["text","image","video"],"features":["json","reasoning","tools"],"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","url":"https://openrouter.ai/moonshotai/kimi-k3","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Morph","tag":"morph/fp8","kind":"host","input":1.875,"output":10.5,"cache_read":0.2175,"context":1048576,"quantization":"fp8"},{"provider":"InferenceNet","tag":"inference-net","kind":"host","input":2.1,"output":10.95,"cache_read":0.23,"context":1048576,"quantization":null},{"provider":"Relace","tag":"relace/fp4","kind":"host","input":2.4,"output":12,"cache_read":0.24,"context":1048576,"quantization":"fp4"},{"provider":"Makora","tag":"makora","kind":"host","input":2.55,"output":12.75,"cache_read":0.256,"context":1048576,"quantization":null},{"provider":"DigitalOcean","tag":"digitalocean","kind":"host","input":2.55,"output":12.95,"cache_read":0.255,"context":1048576,"quantization":null},{"provider":"Sail Research","tag":"sail-research/fp4","kind":"host","input":2.6481,"output":13.2827,"cache_read":0.3026,"context":1048576,"quantization":"fp4"},{"provider":"Phala","tag":"phala","kind":"host","input":2.7,"output":13.5,"cache_read":0.27,"context":1048576,"quantization":null},{"provider":"Wafer","tag":"wafer","kind":"host","input":3,"output":12.75,"cache_read":0.3,"context":1048576,"quantization":null},{"provider":"DeepInfra","tag":"deepinfra/bf16","kind":"host","input":2.85,"output":14.25,"cache_read":0.285,"context":1048576,"quantization":"bf16"},{"provider":"Chutes","tag":"chutes/mxfp4","kind":"host","input":3,"output":15,"cache_read":0.3,"context":1048576,"quantization":"mxfp4"},{"provider":"Parasail","tag":"parasail/fp4","kind":"host","input":3,"output":15,"cache_read":0.3,"context":1048576,"quantization":"fp4"},{"provider":"Modal","tag":"modal/mxfp4","kind":"host","input":3,"output":15,"cache_read":0.3,"context":1048576,"quantization":"mxfp4"},{"provider":"Together","tag":"together","kind":"host","input":3,"output":15,"cache_read":0.3,"context":1048576,"quantization":null},{"provider":"Fireworks","tag":"fireworks","kind":"host","input":3,"output":15,"cache_read":0.3,"context":1048576,"quantization":null},{"provider":"BaseTen","tag":"baseten/fp8","kind":"host","input":3,"output":15,"cache_read":0.3,"context":1048576,"quantization":"fp8"},{"provider":"Moonshot AI","tag":"moonshotai/mxfp4","kind":"host","input":3,"output":15,"cache_read":0.3,"context":1048576,"quantization":"mxfp4"},{"provider":"Alibaba","tag":"alibaba","kind":"host","input":3.45,"output":17.25,"cache_read":0.345,"context":1048576,"quantization":null}]},{"id":"openai/gpt-5.4","slug":"gpt-5-4","name":"GPT-5.4","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":2.5,"output":15,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.25,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","url":"https://openrouter.ai/openai/gpt-5.4","direct_input":1.25,"direct_output":7.5,"direct_cache_read":0.125,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":1.25,"output":7.5,"cache_read":0.125,"context":1050000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":2.5,"output":15,"cache_read":0.25,"context":1050000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/us-east-1","kind":"cloud","input":2.75,"output":16.5,"cache_read":0.275,"context":1050000,"quantization":null}]},{"id":"anthropic/claude-sonnet-4","slug":"claude-sonnet-4","name":"Claude Sonnet 4","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":3,"output":15,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.3,"max_output":64000,"speed":null,"modalities":["image","text","file"],"features":["reasoning","tools"],"description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","url":"https://openrouter.ai/anthropic/claude-sonnet-4","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Amazon Bedrock","tag":"amazon-bedrock/eu-west-1","kind":"cloud","input":3,"output":15,"cache_read":0.3,"context":200000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":3,"output":15,"cache_read":0.3,"context":1000000,"quantization":null}]},{"id":"anthropic/claude-sonnet-4.5","slug":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":3,"output":15,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.3,"max_output":64000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","url":"https://openrouter.ai/anthropic/claude-sonnet-4.5","direct_input":3,"direct_output":15,"direct_cache_read":0.3,"direct_source":"anthropic","routes":[{"provider":"Claude Platform on AWS","tag":"claude-on-aws","kind":"cloud","input":3,"output":15,"cache_read":0.3,"context":1000000,"quantization":null},{"provider":"Azure","tag":"azure/global","kind":"cloud","input":3,"output":15,"cache_read":0.3,"context":200000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":3,"output":15,"cache_read":0.3,"context":1000000,"quantization":null},{"provider":"Anthropic","tag":"anthropic","kind":"official","input":3,"output":15,"cache_read":0.3,"context":1000000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":3,"output":15,"cache_read":0.3,"context":1000000,"quantization":null}]},{"id":"anthropic/claude-sonnet-4.6","slug":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":3,"output":15,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.3,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","url":"https://openrouter.ai/anthropic/claude-sonnet-4.6","direct_input":3,"direct_output":15,"direct_cache_read":0.3,"direct_source":"anthropic","routes":[{"provider":"Claude Platform on AWS","tag":"claude-on-aws","kind":"cloud","input":3,"output":15,"cache_read":0.3,"context":1000000,"quantization":null},{"provider":"Azure","tag":"azure/global","kind":"cloud","input":3,"output":15,"cache_read":0.3,"context":1000000,"quantization":null},{"provider":"Anthropic","tag":"anthropic","kind":"official","input":3,"output":15,"cache_read":0.3,"context":1000000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/global","kind":"cloud","input":3,"output":15,"cache_read":0.3,"context":1000000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":3,"output":15,"cache_read":0.3,"context":1000000,"quantization":null}]},{"id":"openai/gpt-4o-2024-05-13","slug":"gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":5,"output":15,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":4096,"speed":null,"modalities":["text","image","file"],"features":["json","tools"],"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","url":"https://openrouter.ai/openai/gpt-4o-2024-05-13","direct_input":5,"direct_output":15,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":5,"output":15,"cache_read":null,"context":128000,"quantization":null},{"provider":"OpenAI","tag":"openai","kind":"official","input":5,"output":15,"cache_read":null,"context":128000,"quantization":null}]},{"id":"anthropic/claude-opus-4.5","slug":"claude-opus-4-5","name":"Claude Opus 4.5","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":5,"output":25,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.5,"max_output":64000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","url":"https://openrouter.ai/anthropic/claude-opus-4.5","direct_input":5,"direct_output":25,"direct_cache_read":0.5,"direct_source":"anthropic","routes":[{"provider":"Claude Platform on AWS","tag":"claude-on-aws","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":200000,"quantization":null},{"provider":"Azure","tag":"azure/global","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":200000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":200000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":200000,"quantization":null},{"provider":"Anthropic","tag":"anthropic","kind":"official","input":5,"output":25,"cache_read":0.5,"context":200000,"quantization":null}]},{"id":"anthropic/claude-opus-4.6","slug":"claude-opus-4-6","name":"Claude Opus 4.6","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":5,"output":25,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.5,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","url":"https://openrouter.ai/anthropic/claude-opus-4.6","direct_input":5,"direct_output":25,"direct_cache_read":0.5,"direct_source":"anthropic","routes":[{"provider":"Claude Platform on AWS","tag":"claude-on-aws","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Azure","tag":"azure/global","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Anthropic","tag":"anthropic","kind":"official","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null}]},{"id":"anthropic/claude-opus-4.7","slug":"claude-opus-4-7","name":"Claude Opus 4.7","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":5,"output":25,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.5,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","url":"https://openrouter.ai/anthropic/claude-opus-4.7","direct_input":5,"direct_output":25,"direct_cache_read":0.5,"direct_source":"anthropic","routes":[{"provider":"Claude Platform on AWS","tag":"claude-on-aws","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Azure","tag":"azure/global","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/global","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Anthropic","tag":"anthropic","kind":"official","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null}]},{"id":"anthropic/claude-opus-4.8","slug":"claude-opus-4-8","name":"Claude Opus 4.8","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":5,"output":25,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.5,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","url":"https://openrouter.ai/anthropic/claude-opus-4.8","direct_input":5,"direct_output":25,"direct_cache_read":0.5,"direct_source":"anthropic","routes":[{"provider":"Claude Platform on AWS","tag":"claude-on-aws","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Azure","tag":"azure/us","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Anthropic","tag":"anthropic","kind":"official","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null}]},{"id":"anthropic/claude-opus-5","slug":"claude-opus-5","name":"Claude Opus 5","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":5,"output":25,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.5,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","url":"https://openrouter.ai/anthropic/claude-opus-5","direct_input":5,"direct_output":25,"direct_cache_read":0.5,"direct_source":"anthropic","routes":[{"provider":"Azure","tag":"azure/us","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Claude Platform on AWS","tag":"claude-on-aws","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null},{"provider":"Anthropic","tag":"anthropic","kind":"official","input":5,"output":25,"cache_read":0.5,"context":1000000,"quantization":null}]},{"id":"openai/gpt-5.5","slug":"gpt-5-5","name":"GPT-5.5","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":5,"output":30,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.5,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","url":"https://openrouter.ai/openai/gpt-5.5","direct_input":2.5,"direct_output":15,"direct_cache_read":0.25,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":2.5,"output":15,"cache_read":0.25,"context":1050000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":5,"output":30,"cache_read":0.5,"context":1050000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock/us-east-1","kind":"cloud","input":5.5,"output":33,"cache_read":0.55,"context":1050000,"quantization":null}]},{"id":"openai/gpt-chat-latest","slug":"gpt-chat-latest","name":"GPT Chat Latest","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":5,"output":30,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.5,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","tools"],"description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","url":"https://openrouter.ai/openai/gpt-chat-latest","direct_input":5,"direct_output":30,"direct_cache_read":0.5,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":5,"output":30,"cache_read":0.5,"context":400000,"quantization":null}]},{"id":"openai/gpt-4-turbo","slug":"gpt-4-turbo","name":"GPT-4 Turbo","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":10,"output":30,"context":128000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"deprecated","cache_read":null,"max_output":4096,"speed":null,"modalities":["text","image"],"features":["json","tools"],"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling. Training data: up to December 2023.","url":"https://openrouter.ai/openai/gpt-4-turbo","direct_input":10,"direct_output":30,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":10,"output":30,"cache_read":null,"context":128000,"quantization":null}]},{"id":"anthropic/claude-fable-5","slug":"claude-fable-5","name":"Claude Fable 5","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":10,"output":50,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":1,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","url":"https://openrouter.ai/anthropic/claude-fable-5","direct_input":10,"direct_output":50,"direct_cache_read":1,"direct_source":"anthropic","routes":[{"provider":"Claude Platform on AWS","tag":"claude-on-aws","kind":"cloud","input":10,"output":50,"cache_read":1,"context":1000000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":10,"output":50,"cache_read":1,"context":1000000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":10,"output":50,"cache_read":1,"context":1000000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":10,"output":50,"cache_read":1,"context":1000000,"quantization":null},{"provider":"Anthropic","tag":"anthropic","kind":"official","input":10,"output":50,"cache_read":1,"context":1000000,"quantization":null}]},{"id":"anthropic/claude-fable-5.1","slug":"claude-fable-5-1","name":"Claude Fable 5.1","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":10,"output":50,"context":1000000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":0.25,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","url":"https://openrouter.ai/anthropic/claude-fable-5.1","direct_input":10,"direct_output":50,"direct_cache_read":0.25,"direct_source":"anthropic","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":10,"output":50,"cache_read":0.25,"context":1000000,"quantization":null},{"provider":"Anthropic","tag":"anthropic","kind":"official","input":10,"output":50,"cache_read":0.25,"context":1000000,"quantization":null},{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":10,"output":50,"cache_read":0.25,"context":1000000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":10,"output":50,"cache_read":0.25,"context":1000000,"quantization":null}]},{"id":"openai/gpt-6-astra","slug":"gpt-6-astra","name":"GPT-6 Astra","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":10,"output":50,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":1,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","url":"https://openrouter.ai/openai/gpt-6-astra","direct_input":5,"direct_output":25,"direct_cache_read":0.5,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":5,"output":25,"cache_read":0.5,"context":1050000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":10,"output":50,"cache_read":1,"context":1050000,"quantization":null}]},{"id":"openai/gpt-6-astra-pro","slug":"gpt-6-astra-pro","name":"GPT-6 Astra Pro","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":10,"output":50,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":1,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-6 Astra Pro is the same underlying model as [GPT-6 Astra](https://openrouter.ai/openai/gpt-6-astra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","url":"https://openrouter.ai/openai/gpt-6-astra-pro","direct_input":5,"direct_output":25,"direct_cache_read":0.5,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":5,"output":25,"cache_read":0.5,"context":1050000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":10,"output":50,"cache_read":1,"context":1050000,"quantization":null}]},{"id":"openai/o1","slug":"o1","name":"o1","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":15,"output":60,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":7.5,"max_output":100000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","url":"https://openrouter.ai/openai/o1","direct_input":15,"direct_output":60,"direct_cache_read":7.5,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":15,"output":60,"cache_read":7.5,"context":200000,"quantization":null}]},{"id":"anthropic/claude-opus-4","slug":"claude-opus-4","name":"Claude Opus 4","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":15,"output":75,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":1.5,"max_output":32000,"speed":null,"modalities":["image","text","file"],"features":["reasoning","tools"],"description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","url":"https://openrouter.ai/anthropic/claude-opus-4","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Google","tag":"google-vertex","kind":"cloud","input":15,"output":75,"cache_read":1.5,"context":200000,"quantization":null}]},{"id":"anthropic/claude-opus-4.1","slug":"claude-opus-4-1","name":"Claude Opus 4.1","vendor":"anthropic","vendor_name":"Anthropic","provider":"OpenRouter","provider_id":"openrouter","family":"claude","input":15,"output":75,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":1.5,"max_output":32000,"speed":null,"modalities":["image","text","file"],"features":["reasoning","tools"],"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","url":"https://openrouter.ai/anthropic/claude-opus-4.1","direct_input":null,"direct_output":null,"direct_cache_read":null,"direct_source":null,"routes":[{"provider":"Amazon Bedrock","tag":"amazon-bedrock","kind":"cloud","input":15,"output":75,"cache_read":1.5,"context":200000,"quantization":null},{"provider":"Google","tag":"google-vertex/global","kind":"cloud","input":15,"output":75,"cache_read":1.5,"context":200000,"quantization":null}]},{"id":"openai/o3-pro","slug":"o3-pro","name":"o3 Pro","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":20,"output":80,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":100000,"speed":null,"modalities":["text","file","image"],"features":["json","reasoning","tools"],"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","url":"https://openrouter.ai/openai/o3-pro","direct_input":20,"direct_output":80,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":20,"output":80,"cache_read":null,"context":200000,"quantization":null}]},{"id":"openai/gpt-4","slug":"gpt-4","name":"GPT-4","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":30,"output":60,"context":8191,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":4096,"speed":null,"modalities":["text"],"features":["json","tools"],"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","url":"https://openrouter.ai/openai/gpt-4","direct_input":30,"direct_output":60,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"Azure","tag":"azure","kind":"cloud","input":30,"output":60,"cache_read":null,"context":8191,"quantization":null},{"provider":"OpenAI","tag":"openai","kind":"official","input":30,"output":60,"cache_read":null,"context":8191,"quantization":null}]},{"id":"openai/gpt-5-pro","slug":"gpt-5-pro","name":"GPT-5 Pro","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":15,"output":120,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":128000,"speed":null,"modalities":["image","text","file"],"features":["json","reasoning","tools"],"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","url":"https://openrouter.ai/openai/gpt-5-pro","direct_input":15,"direct_output":120,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":15,"output":120,"cache_read":null,"context":400000,"quantization":null}]},{"id":"openai/gpt-5.2-pro","slug":"gpt-5-2-pro","name":"GPT-5.2 Pro","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":21,"output":168,"context":400000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":128000,"speed":null,"modalities":["image","text","file"],"features":["json","reasoning","tools"],"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","url":"https://openrouter.ai/openai/gpt-5.2-pro","direct_input":21,"direct_output":168,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":21,"output":168,"cache_read":null,"context":400000,"quantization":null}]},{"id":"openai/gpt-5.4-pro","slug":"gpt-5-4-pro","name":"GPT-5.4 Pro","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":30,"output":180,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":128000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning","tools"],"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","url":"https://openrouter.ai/openai/gpt-5.4-pro","direct_input":15,"direct_output":90,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":15,"output":90,"cache_read":null,"context":1050000,"quantization":null},{"provider":"Azure","tag":"azure","kind":"cloud","input":30,"output":180,"cache_read":null,"context":1050000,"quantization":null}]},{"id":"openai/gpt-5.5-pro","slug":"gpt-5-5-pro","name":"GPT-5.5 Pro","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":30,"output":180,"context":1050000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":128000,"speed":null,"modalities":["file","image","text"],"features":["json","reasoning","tools"],"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","url":"https://openrouter.ai/openai/gpt-5.5-pro","direct_input":15,"direct_output":90,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai/flex","kind":"official","input":15,"output":90,"cache_read":null,"context":1050000,"quantization":null}]},{"id":"openai/o1-pro","slug":"o1-pro","name":"o1-pro","vendor":"openai","vendor_name":"OpenAI","provider":"OpenRouter","provider_id":"openrouter","family":"gpt","input":150,"output":600,"context":200000,"source":"openrouter","source_url":"https://openrouter.ai/api/v1/models","status":"ga","cache_read":null,"max_output":100000,"speed":null,"modalities":["text","image","file"],"features":["json","reasoning"],"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","url":"https://openrouter.ai/openai/o1-pro","direct_input":150,"direct_output":600,"direct_cache_read":null,"direct_source":"openai","routes":[{"provider":"OpenAI","tag":"openai","kind":"official","input":150,"output":600,"cache_read":null,"context":200000,"quantization":null}]}]}