{"meta":{"count":340,"total":340,"currency":"USD","unit":"per 1,000,000 tokens","blendWeights":{"input":0.75,"output":0.25},"fetchedAt":"2026-09-14T18:22:24.027Z","stale":false,"license":"CC-BY-4.0","attribution":"https://www.getsnippets.ai"},"data":[{"id":"inference-net/schematron-v2-turbo","slug":"inference-net-schematron-v2-turbo","name":"Inference.net: Schematron V2 Turbo","shortName":"Schematron V2 Turbo","providerId":"inference-net","description":"Schematron V2 Turbo is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes throughput for high-volume extraction workloads. Extraction instructions must be supplied through a JSON schema in response_format rather...","contextLength":128000,"maxOutputTokens":8192,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.03,"outputPerMTok":0.15,"cachedInputPerMTok":0.03,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/inference-net/schematron-v2-turbo"},"blendedPerMTok":0.06,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"inference-net/schematron-v2-granite-4.0-h-micro","benchmarks":null,"releasedAt":"2026-09-12T01:35:49.000Z","isFree":false},{"id":"inference-net/schematron-v2-small","slug":"inference-net-schematron-v2-small","name":"Inference.net: Schematron V2 Small","shortName":"Schematron V2 Small","providerId":"inference-net","description":"Schematron V2 Small is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes extraction quality for complex schemas and long pages. Extraction instructions must be supplied through a JSON schema...","contextLength":128000,"maxOutputTokens":4096,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.049999999999999996,"outputPerMTok":0.22999999999999998,"cachedInputPerMTok":0.049999999999999996,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/inference-net/schematron-v2-small"},"blendedPerMTok":0.095,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"inference-net/schematron-v2-llama-3.2-3b","benchmarks":null,"releasedAt":"2026-09-12T01:35:33.000Z","isFree":false},{"id":"sakana/fugu-ultra-v2","slug":"sakana-fugu-ultra-v2","name":"Sakana: Fugu Ultra v2","shortName":"Fugu Ultra v2","providerId":"sakana","description":"Fugu Ultra v2 is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to...","contextLength":1000000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":5,"outputPerMTok":30,"cachedInputPerMTok":0.5,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/sakana/fugu-ultra-v2"},"blendedPerMTok":11.25,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2026-08-28","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-09-11T05:43:03.000Z","isFree":false},{"id":"sakana/fugu-max","slug":"sakana-fugu-max","name":"Sakana: Fugu Max","shortName":"Fugu Max","providerId":"sakana","description":"Fugu Max is the cost-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","contextLength":1000000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":6,"cachedInputPerMTok":0.25,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/sakana/fugu-max"},"blendedPerMTok":3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-09-11T05:32:51.000Z","isFree":false},{"id":"inclusionai/ling-3.0-flash-vl","slug":"inclusionai-ling-3-0-flash-vl","name":"inclusionAI: Ling 3.0 Flash VL","shortName":"Ling 3.0 Flash VL","providerId":"inclusionai","description":"Ling 3.0 Flash VL builds on Ling 3.0 Flash (124B total / 5.5B active MoE from InclusionAI), further strengthening its language capabilities while adding native visual perception and advanced visual...","contextLength":131072,"maxOutputTokens":32768,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.06,"outputPerMTok":0.18,"cachedInputPerMTok":0.012,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-vl"},"blendedPerMTok":0.09,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"inclusionAI/Ling-3.0-flash-VL","benchmarks":{"intelligence_index":24.8,"coding_index":57,"agentic_index":30},"releasedAt":"2026-09-10T16:01:54.000Z","isFree":true},{"id":"deepseek/deepseek-v4.1-flash","slug":"deepseek-deepseek-v4-1-flash","name":"DeepSeek: DeepSeek V4.1 Flash","shortName":"DeepSeek V4.1 Flash","providerId":"deepseek","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","contextLength":1048576,"maxOutputTokens":384000,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":0.6,"cachedInputPerMTok":0.003,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4.1-flash"},"blendedPerMTok":0.26249999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V4.1-Flash","benchmarks":{"intelligence_index":39.5},"releasedAt":"2026-09-10T06:21:25.000Z","isFree":false},{"id":"inception/mercury-2.5","slug":"inception-mercury-2-5","name":"Inception: Mercury 2.5","shortName":"Mercury 2.5","providerId":"inception","description":"Mercury 2.5 is the fastest reasoning LLM, and the latest diffusion LLM (dLLM) from Inception. Instead of generating tokens sequentially, Mercury 2.5 produces and refines multiple tokens in parallel, achieving...","contextLength":260000,"maxOutputTokens":65536,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.04,"outputPerMTok":0.15,"cachedInputPerMTok":0.004,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/inception/mercury-2.5"},"blendedPerMTok":0.0675,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-09-08T18:28:57.000Z","isFree":false},{"id":"nex-agi/nex-n2.5-mini","slug":"nex-agi-nex-n2-5-mini","name":"Nex AGI: Nex-N2.5-Mini (free)","shortName":"Nex-N2.5-Mini (free)","providerId":"nex-agi","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0,"outputPerMTok":0,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/nex-agi/nex-n2.5-mini"},"blendedPerMTok":0,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"nex-agi/Nex-N2.5-mini","benchmarks":null,"releasedAt":"2026-09-08T17:54:21.000Z","isFree":true},{"id":"nex-agi/nex-n2.5-pro","slug":"nex-agi-nex-n2-5-pro","name":"Nex AGI: Nex-N2.5-Pro (free)","shortName":"Nex-N2.5-Pro (free)","providerId":"nex-agi","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0,"outputPerMTok":0,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/nex-agi/nex-n2.5-pro"},"blendedPerMTok":0,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"nex-agi/Nex-N2.5-Pro","benchmarks":null,"releasedAt":"2026-09-08T17:54:10.000Z","isFree":true},{"id":"openai/gpt-6-astra","slug":"openai-gpt-6-astra","name":"OpenAI: GPT-6 Astra","shortName":"GPT-6 Astra","providerId":"openai","description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":10,"outputPerMTok":50,"cachedInputPerMTok":1,"cacheWritePerMTok":12.5,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-6-astra"},"blendedPerMTok":20,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":52.8,"coding_index":76.9,"agentic_index":51.5},"releasedAt":"2026-09-04T20:13:58.000Z","isFree":false},{"id":"openai/gpt-6-astra-pro","slug":"openai-gpt-6-astra-pro","name":"OpenAI: GPT-6 Astra Pro","shortName":"GPT-6 Astra Pro","providerId":"openai","description":"GPT-6 Astra Pro is the same underlying model as [GPT-6 Astra](https://openrouter.ai/openai/gpt-6-astra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":10,"outputPerMTok":50,"cachedInputPerMTok":1,"cacheWritePerMTok":12.5,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-6-astra-pro"},"blendedPerMTok":20,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-09-04T20:13:55.000Z","isFree":false},{"id":"inclusionai/ling-3.0-flash-sante","slug":"inclusionai-ling-3-0-flash-sante","name":"inclusionAI: Ling 3.0 Flash Sante (free)","shortName":"Ling 3.0 Flash Sante (free)","providerId":"inclusionai","description":"Ling 3.0 Flash Sante is a health and medicine-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for...","contextLength":262144,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0,"outputPerMTok":0,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-sante"},"blendedPerMTok":0,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-09-04T18:19:06.000Z","isFree":true},{"id":"qwen/qwen3.8-max-0902","slug":"qwen-qwen3-8-max-0902","name":"Qwen: Qwen3.8 Max (0902)","shortName":"Qwen3.8 Max (0902)","providerId":"qwen","description":"Qwen3.8 Max 0902 is an updated snapshot of Qwen3.8 Max from Alibaba's Qwen team. It is a 2.4-trillion-parameter mixture-of-experts model that accepts text, image, and video input and returns text,...","contextLength":1000000,"maxOutputTokens":131072,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":6,"cachedInputPerMTok":0.25,"cacheWritePerMTok":2.5,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-max-0902"},"blendedPerMTok":3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":40.3,"coding_index":71.8,"agentic_index":49.6},"releasedAt":"2026-09-03T21:08:24.000Z","isFree":false},{"id":"meta/muse-spark-1.3-contributor","slug":"meta-muse-spark-1-3-contributor","name":"Meta: Muse Spark 1.3 Contributor","shortName":"Muse Spark 1.3 Contributor","providerId":"meta","description":"Muse Spark 1.3 Contributor is the cost-efficient contributor tier of Meta’s multimodal reasoning model for experimentation, learning, and early-stage agentic, multi-agent, and coding workflows. It is designed to track information...","contextLength":1048576,"maxOutputTokens":943718,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.19999999999999998,"cachedInputPerMTok":0.002,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.0025,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.3-contributor"},"blendedPerMTok":0.125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-09-02T20:38:39.000Z","isFree":false},{"id":"meta/muse-spark-1.3","slug":"meta-muse-spark-1-3","name":"Meta: Muse Spark 1.3","shortName":"Muse Spark 1.3","providerId":"meta","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It is designed to keep track of information across extended tasks, work through...","contextLength":1048576,"maxOutputTokens":943718,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":4.25,"cachedInputPerMTok":0.15,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.0025,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.3"},"blendedPerMTok":2,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-09-02T19:45:59.000Z","isFree":false},{"id":"google/gemini-3.8-flash","slug":"google-gemini-3-8-flash","name":"Google: Gemini 3.8 Flash","shortName":"Gemini 3.8 Flash","providerId":"google","description":"Gemini 3.8 Flash is Google's most intelligent Flash model with significant gains from 3.7 Flash across software engineering, agentic tasks, and multi-step reasoning.","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":0.75,"outputPerMTok":3.75,"cachedInputPerMTok":0.075,"cacheWritePerMTok":0.0416666666666667,"perRequest":null,"perImage":7.5e-7,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.8-flash"},"blendedPerMTok":1.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":41.2,"coding_index":76.3,"agentic_index":41.1},"releasedAt":"2026-09-02T15:14:16.000Z","isFree":false},{"id":"anthropic/claude-fable-5.1","slug":"anthropic-claude-fable-5-1","name":"Anthropic: Claude Fable 5.1","shortName":"Claude Fable 5.1","providerId":"anthropic","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","contextLength":1000000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":10,"outputPerMTok":50,"cachedInputPerMTok":0.25,"cacheWritePerMTok":12.5,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-fable-5.1"},"blendedPerMTok":20,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":53.4,"coding_index":81.6,"agentic_index":58},"releasedAt":"2026-09-01T18:03:58.000Z","isFree":false},{"id":"ibm-granite/granite-4.2-8b","slug":"ibm-granite-granite-4-2-8b","name":"IBM: Granite 4.2 8B","shortName":"Granite 4.2 8B","providerId":"ibm-granite","description":"Granite 4.2 8B is a dense reasoning model from IBM. It is suited for mathematics, code generation, multilingual dialogue, and agentic workflows that need multi-step reasoning. It supports full, low-effort,...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.06,"outputPerMTok":0.25,"cachedInputPerMTok":0.015,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/ibm-granite/granite-4.2-8b"},"blendedPerMTok":0.1075,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"ibm-granite/granite-4.2-8b","benchmarks":{"intelligence_index":11.8,"coding_index":22.4,"agentic_index":3.7},"releasedAt":"2026-08-31T20:06:20.000Z","isFree":false},{"id":"tencent/hy4-preview","slug":"tencent-hy4-preview","name":"Tencent: Hy4 preview","shortName":"Hy4 preview","providerId":"tencent","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","contextLength":1048576,"maxOutputTokens":64000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.834,"outputPerMTok":2.501,"cachedInputPerMTok":0.041999999999999996,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/tencent/hy4-preview"},"blendedPerMTok":1.25075,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"tencent/Hy4-preview","benchmarks":null,"releasedAt":"2026-08-28T06:09:35.000Z","isFree":false},{"id":"inclusionai/ling-3.0-flash-fin","slug":"inclusionai-ling-3-0-flash-fin","name":"inclusionAI: Ling 3.0 Flash Fin","shortName":"Ling 3.0 Flash Fin","providerId":"inclusionai","description":"Ling 3.0 Flash Fin is a finance-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for real-world investment...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.06,"outputPerMTok":0.18,"cachedInputPerMTok":0.012,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-fin"},"blendedPerMTok":0.09,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-08-27T15:58:10.000Z","isFree":true},{"id":"qwen/qwen3.8-flash","slug":"qwen-qwen3-8-flash","name":"Qwen: Qwen3.8 Flash","shortName":"Qwen3.8 Flash","providerId":"qwen","description":"Qwen3.8 Flash is a multimodal reasoning model from Alibaba. It is suited for coding assistance, agentic workflows, visual understanding, document and codebase analysis, desktop interaction, chart analysis, and long-video analysis.","contextLength":1000000,"maxOutputTokens":131072,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":0.47,"cachedInputPerMTok":0.016,"cacheWritePerMTok":0.19999999999999998,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-flash"},"blendedPerMTok":0.22999999999999998,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3.8-Flash-Next","benchmarks":null,"releasedAt":"2026-08-26T19:37:40.000Z","isFree":false},{"id":"z-ai/glm-5.3-flash","slug":"z-ai-glm-5-3-flash","name":"Z.ai: GLM 5.3 Flash","shortName":"GLM 5.3 Flash","providerId":"z-ai","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","contextLength":1310720,"maxOutputTokens":131072,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":0.5,"cachedInputPerMTok":0.03,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3-flash"},"blendedPerMTok":0.2375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"zai-org/GLM-5.3-Flash","benchmarks":{"intelligence_index":41.9,"coding_index":71.5,"agentic_index":51.2},"releasedAt":"2026-08-26T13:59:01.000Z","isFree":false},{"id":"meta/muse-spark-1.2-contributor","slug":"meta-muse-spark-1-2-contributor","name":"Meta: Muse Spark 1.2 Contributor","shortName":"Muse Spark 1.2 Contributor","providerId":"meta","description":"Muse Spark 1.2 contributor tier is a reasoning model from Meta designed for developers who want to start building at an even lower cost. It’s meaningfully cheaper than Muse Spark...","contextLength":1048576,"maxOutputTokens":943718,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.19999999999999998,"cachedInputPerMTok":0.002,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.0025,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.2-contributor"},"blendedPerMTok":0.125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-08-21T18:21:16.000Z","isFree":false},{"id":"deepseek/deepseek-v4-flash-vision-exp","slug":"deepseek-deepseek-v4-flash-vision-exp","name":"DeepSeek: DeepSeek V4 Flash Vision Exp","shortName":"DeepSeek V4 Flash Vision Exp","providerId":"deepseek","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","contextLength":1048576,"maxOutputTokens":943718,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.22,"outputPerMTok":0.66,"cachedInputPerMTok":0.007,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-flash-vision-exp"},"blendedPerMTok":0.33,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","benchmarks":null,"releasedAt":"2026-08-21T11:26:03.000Z","isFree":false},{"id":"tencent/hy-mt2-1.8b","slug":"tencent-hy-mt2-1-8b","name":"Tencent: Hy-MT2-1.8B","shortName":"Hy-MT2-1.8B","providerId":"tencent","description":"Hy-MT2-1.8B is a compact 1.8B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided...","contextLength":8192,"maxOutputTokens":4096,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.044,"outputPerMTok":0.17700000000000002,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/tencent/hy-mt2-1.8b"},"blendedPerMTok":0.07725000000000001,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"tencent/Hy-MT2-1.8B","benchmarks":null,"releasedAt":"2026-08-20T13:13:01.000Z","isFree":false},{"id":"tencent/hy-mt2-30b-a3b","slug":"tencent-hy-mt2-30b-a3b","name":"Tencent: Hy-MT2-30B-A3B","shortName":"Hy-MT2-30B-A3B","providerId":"tencent","description":"Hy-MT2-30B-A3B is Tencent's flagship translation model in the Hy-MT2 family. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and...","contextLength":8192,"maxOutputTokens":4096,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.074,"outputPerMTok":0.295,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/tencent/hy-mt2-30b-a3b"},"blendedPerMTok":0.12924999999999998,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"tencent/Hy-MT2-30B-A3B","benchmarks":null,"releasedAt":"2026-08-20T13:12:41.000Z","isFree":false},{"id":"tencent/hy-mt2-7b","slug":"tencent-hy-mt2-7b","name":"Tencent: Hy-MT2-7B","shortName":"Hy-MT2-7B","providerId":"tencent","description":"Hy-MT2-7B is a 7B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided translation.","contextLength":8192,"maxOutputTokens":4096,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.074,"outputPerMTok":0.295,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/tencent/hy-mt2-7b"},"blendedPerMTok":0.12924999999999998,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"tencent/Hy-MT2-7B","benchmarks":null,"releasedAt":"2026-08-19T14:13:17.000Z","isFree":false},{"id":"z-ai/glm-5.3","slug":"z-ai-glm-5-3","name":"Z.ai: GLM 5.3","shortName":"GLM 5.3","providerId":"z-ai","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","contextLength":1310720,"maxOutputTokens":943717,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":1.4,"outputPerMTok":4.4,"cachedInputPerMTok":0.26,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3"},"blendedPerMTok":2.15,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"zai-org/GLM-5.3","benchmarks":{"intelligence_index":44.9,"coding_index":74.8,"agentic_index":53.4},"releasedAt":"2026-08-18T20:57:35.000Z","isFree":false},{"id":"qwen/qwen3.8-27b","slug":"qwen-qwen3-8-27b","name":"Qwen: Qwen3.8 27B","shortName":"Qwen3.8 27B","providerId":"qwen","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","contextLength":1000000,"maxOutputTokens":131072,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.21400000000000002,"outputPerMTok":2.5500000000000003,"cachedInputPerMTok":0.15,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-27b"},"blendedPerMTok":0.798,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3.8-27B","benchmarks":{"intelligence_index":33.9,"coding_index":68.1,"agentic_index":46.5},"releasedAt":"2026-08-14T15:55:10.000Z","isFree":false},{"id":"dots-studio/dots-3-note-preview","slug":"dots-studio-dots-3-note-preview","name":"Dots Studio: Dots3-Note Preview (free)","shortName":"Dots3-Note Preview (free)","providerId":"dots-studio","description":"Dots3-Note Preview is an open-weight mixture-of-experts model from Dots Studio, with 16B active parameters out of 280B total. It is the lightest model in the Dots 3 family and is...","contextLength":512000,"maxOutputTokens":460800,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0,"outputPerMTok":0,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/dots-studio/dots-3-note-preview"},"blendedPerMTok":0,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-08-14T04:06:01.000Z","isFree":true},{"id":"google/gemini-3.7-flash","slug":"google-gemini-3-7-flash","name":"Google: Gemini 3.7 Flash","shortName":"Gemini 3.7 Flash","providerId":"google","description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":0.75,"outputPerMTok":3.75,"cachedInputPerMTok":0.075,"cacheWritePerMTok":0.0416666666666667,"perRequest":null,"perImage":7.5e-7,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.7-flash"},"blendedPerMTok":1.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":39.4,"coding_index":76.1,"agentic_index":36.4},"releasedAt":"2026-08-13T17:03:01.000Z","isFree":false},{"id":"bytedance-seed/seed-2-1-turbo","slug":"bytedance-seed-seed-2-1-turbo","name":"ByteDance Seed: Seed 2.1 Turbo","shortName":"Seed 2.1 Turbo","providerId":"bytedance-seed","description":"Seed 2.1 Turbo is a multimodal model from ByteDance Seed for coding and long-horizon agent workflows. It is suited for end-to-end software delivery, multi-step task execution, and understanding visual and...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.5,"outputPerMTok":2.5,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/bytedance-seed/seed-2-1-turbo"},"blendedPerMTok":1,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-08-12T16:29:36.000Z","isFree":false},{"id":"qwen/qwen3.8-2.4t-a95b","slug":"qwen-qwen3-8-2-4t-a95b","name":"Qwen: Qwen3.8 2.4T A95B","shortName":"Qwen3.8 2.4T A95B","providerId":"qwen","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","contextLength":1048576,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":6,"cachedInputPerMTok":0.25,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-2.4t-a95b"},"blendedPerMTok":3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3.8-2.4T-A95B","benchmarks":{"intelligence_index":40,"coding_index":71.9,"agentic_index":50.4},"releasedAt":"2026-08-12T16:21:42.000Z","isFree":false},{"id":"bytedance-seed/seed-2.0-code","slug":"bytedance-seed-seed-2-0-code","name":"ByteDance Seed: Seed-2.0-Code","shortName":"Seed-2.0-Code","providerId":"bytedance-seed","description":"Seed 2.0 Code is a model from ByteDance Seed optimized for agentic coding. It is suited for frontend development, multilingual programming tasks, and coding-agent workflows in tools such as Claude...","contextLength":262144,"maxOutputTokens":131072,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.5,"outputPerMTok":3,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/bytedance-seed/seed-2.0-code"},"blendedPerMTok":1.125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-08-12T16:05:01.000Z","isFree":false},{"id":"deepseek/deepseek-v4-pro-0813","slug":"deepseek-deepseek-v4-pro-0813","name":"DeepSeek: DeepSeek V4 Pro 0813","shortName":"DeepSeek V4 Pro 0813","providerId":"deepseek","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","contextLength":1048576,"maxOutputTokens":384000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.9833999999999999,"outputPerMTok":2.9502,"cachedInputPerMTok":0.03278,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-pro-0813"},"blendedPerMTok":1.4750999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V4-Pro-0813","benchmarks":{"intelligence_index":36.3,"coding_index":68.8,"agentic_index":42.3},"releasedAt":"2026-08-12T15:42:44.000Z","isFree":false},{"id":"x-ai/grok-4.6","slug":"x-ai-grok-4-6","name":"SpaceXAI: Grok 4.6","shortName":"Grok 4.6","providerId":"x-ai","description":"Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","contextLength":500000,"maxOutputTokens":450000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":6,"cachedInputPerMTok":0.5,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.005,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/x-ai/grok-4.6"},"blendedPerMTok":3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":44.4,"coding_index":76.8,"agentic_index":53.4},"releasedAt":"2026-08-12T15:35:57.000Z","isFree":false},{"id":"liquid/lfm-2.5-2.6b","slug":"liquid-lfm-2-5-2-6b","name":"LiquidAI: LFM2.5-2.6B (free)","shortName":"LFM2.5-2.6B (free)","providerId":"liquid","description":"LFM2.5-2.6B is a compact reasoning model from Liquid AI. It is suited for agent workflows, data extraction, RAG, and long-context processing. Liquid advises against using it for agentic coding or...","contextLength":65536,"maxOutputTokens":8192,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0,"outputPerMTok":0,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/liquid/lfm-2.5-2.6b"},"blendedPerMTok":0,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"LiquidAI/LFM2.5-2.6B","benchmarks":null,"releasedAt":"2026-08-11T17:48:39.000Z","isFree":true},{"id":"nvidia/nemotron-3.5-lightning","slug":"nvidia-nemotron-3-5-lightning","name":"NVIDIA: Nemotron 3.5 Lightning","shortName":"Nemotron 3.5 Lightning","providerId":"nvidia","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","contextLength":262144,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.08,"outputPerMTok":0.19999999999999998,"cachedInputPerMTok":0.04,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3.5-lightning"},"blendedPerMTok":0.10999999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","benchmarks":{"intelligence_index":13.6,"coding_index":26.8,"agentic_index":6.1},"releasedAt":"2026-08-11T12:52:31.000Z","isFree":true},{"id":"sakana/sakana-namazu","slug":"sakana-sakana-namazu","name":"Sakana: Sakana Namazu","shortName":"Sakana Namazu","providerId":"sakana","description":"Sakana Namazu is a Japanese-specialized reasoning model from Sakana AI, based on Kimi K2.6 with additional training for Japanese language and business contexts. It is suited for Japanese instruction following,...","contextLength":262144,"maxOutputTokens":65536,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.95,"outputPerMTok":4,"cachedInputPerMTok":0.15,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.007,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/sakana/sakana-namazu"},"blendedPerMTok":1.7125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-08-11T01:02:09.000Z","isFree":false},{"id":"upstage/solar-pro4","slug":"upstage-solar-pro4","name":"Upstage: Solar Pro 4","shortName":"Solar Pro 4","providerId":"upstage","description":"Solar Pro 4 is Upstage's cost-efficient large language model, featuring a 524K context window. It is built for long-horizon tasks and agentic workflows, with strong capabilities in office productivity, document-intensive...","contextLength":524288,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.09,"outputPerMTok":0.36,"cachedInputPerMTok":0.018,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/upstage/solar-pro4"},"blendedPerMTok":0.1575,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"coding_index":52.7},"releasedAt":"2026-08-10T14:20:36.000Z","isFree":false},{"id":"meta/muse-glimmer-30b","slug":"meta-muse-glimmer-30b","name":"Meta: Muse Glimmer 30B","shortName":"Muse Glimmer 30B","providerId":"meta","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.35,"outputPerMTok":1.5,"cachedInputPerMTok":0.04,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta/muse-glimmer-30b"},"blendedPerMTok":0.6375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"meta-models/Muse-Glimmer-30B","benchmarks":null,"releasedAt":"2026-08-09T19:06:34.000Z","isFree":false},{"id":"meta/muse-spark-1.2","slug":"meta-muse-spark-1-2","name":"Meta: Muse Spark 1.2","shortName":"Muse Spark 1.2","providerId":"meta","description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...","contextLength":1048576,"maxOutputTokens":943718,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":4.25,"cachedInputPerMTok":0.15,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.0025,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.2"},"blendedPerMTok":2,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":39.8,"coding_index":72.2,"agentic_index":44},"releasedAt":"2026-08-05T19:48:07.000Z","isFree":false},{"id":"deepseek/deepseek-v4-flash-0731","slug":"deepseek-deepseek-v4-flash-0731","name":"DeepSeek: DeepSeek V4 Flash 0731","shortName":"DeepSeek V4 Flash 0731","providerId":"deepseek","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","contextLength":1310720,"maxOutputTokens":943718,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.06,"outputPerMTok":0.12,"cachedInputPerMTok":0.012,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-flash-0731"},"blendedPerMTok":0.075,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V4-Flash-0731","benchmarks":{"intelligence_index":34.5,"coding_index":69.1,"agentic_index":41.7},"releasedAt":"2026-07-31T06:21:48.000Z","isFree":false},{"id":"thinkingmachines/inkling-small","slug":"thinkingmachines-inkling-small","name":"Thinking Machines: Inkling Small","shortName":"Inkling Small","providerId":"thinkingmachines","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","contextLength":1048576,"maxOutputTokens":262144,"inputModalities":["text","image","audio"],"outputModalities":["text"],"price":{"inputPerMTok":0.44999999999999996,"outputPerMTok":1.2,"cachedInputPerMTok":0.09999999999999999,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling-small"},"blendedPerMTok":0.6375,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"thinkingmachines/Inkling-Small","benchmarks":{"intelligence_index":26.1,"coding_index":52.9,"agentic_index":25},"releasedAt":"2026-07-30T20:25:17.000Z","isFree":true},{"id":"qwen/qwen3.7-flash","slug":"qwen-qwen3-7-flash","name":"Qwen: Qwen3.7 Flash","shortName":"Qwen3.7 Flash","providerId":"qwen","description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","contextLength":1000000,"maxOutputTokens":65536,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.03,"outputPerMTok":0.13,"cachedInputPerMTok":0.006,"cacheWritePerMTok":0.038000000000000006,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-flash"},"blendedPerMTok":0.055,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-07-27T22:16:01.000Z","isFree":false},{"id":"anthropic/claude-opus-5","slug":"anthropic-claude-opus-5","name":"Claude Opus 5","shortName":"Claude Opus 5","providerId":"anthropic","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","contextLength":1000000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":5,"outputPerMTok":25,"cachedInputPerMTok":0.5,"cacheWritePerMTok":6.25,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5"},"blendedPerMTok":10,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":50.7,"coding_index":78,"agentic_index":56.2},"releasedAt":"2026-07-24T17:02:24.000Z","isFree":false},{"id":"inclusionai/ling-3.0-flash","slug":"inclusionai-ling-3-0-flash","name":"inclusionAI: Ling 3.0 Flash","shortName":"Ling 3.0 Flash","providerId":"inclusionai","description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","contextLength":262144,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.020999999999999998,"outputPerMTok":0.063,"cachedInputPerMTok":0.004200000000000001,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash"},"blendedPerMTok":0.0315,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"inclusionAI/Ling-3.0-flash","benchmarks":{"coding_index":50.6,"agentic_index":21},"releasedAt":"2026-07-23T14:56:20.000Z","isFree":false},{"id":"poolside/laguna-s-2.1","slug":"poolside-laguna-s-2-1","name":"Poolside: Laguna S 2.1","shortName":"Laguna S 2.1","providerId":"poolside","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","contextLength":1048576,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.09,"outputPerMTok":0.18,"cachedInputPerMTok":0.009,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/poolside/laguna-s-2.1"},"blendedPerMTok":0.1125,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"poolside/Laguna-S-2.1","benchmarks":null,"releasedAt":"2026-07-21T16:51:23.000Z","isFree":true},{"id":"google/gemini-3.6-flash","slug":"google-gemini-3-6-flash","name":"Google: Gemini 3.6 Flash","shortName":"Gemini 3.6 Flash","providerId":"google","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":0.75,"outputPerMTok":3.75,"cachedInputPerMTok":0.075,"cacheWritePerMTok":0.0416666666666667,"perRequest":null,"perImage":7.5e-7,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.6-flash"},"blendedPerMTok":1.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":34.3,"coding_index":69.2,"agentic_index":30.2},"releasedAt":"2026-07-21T15:12:13.000Z","isFree":false},{"id":"google/gemini-3.5-flash-lite","slug":"google-gemini-3-5-flash-lite","name":"Google: Gemini 3.5 Flash Lite","shortName":"Gemini 3.5 Flash Lite","providerId":"google","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":2.5,"cachedInputPerMTok":0.03,"cacheWritePerMTok":0.0833333333333333,"perRequest":null,"perImage":3e-7,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash-lite"},"blendedPerMTok":0.85,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":22.7,"coding_index":49.3,"agentic_index":15.9},"releasedAt":"2026-07-21T15:12:06.000Z","isFree":false},{"id":"meituan/longcat-2.0","slug":"meituan-longcat-2-0","name":"Meituan: LongCat 2.0","shortName":"LongCat 2.0","providerId":"meituan","description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","contextLength":1048756,"maxOutputTokens":262144,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":1.2,"cachedInputPerMTok":0.006,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meituan/longcat-2.0"},"blendedPerMTok":0.5249999999999999,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"meituan-longcat/LongCat-2.0","benchmarks":{"intelligence_index":19.7,"coding_index":45.3,"agentic_index":15.9},"releasedAt":"2026-07-20T13:37:38.000Z","isFree":false},{"id":"thinkingmachines/inkling","slug":"thinkingmachines-inkling","name":"Thinking Machines: Inkling","shortName":"Inkling","providerId":"thinkingmachines","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","contextLength":1048576,"maxOutputTokens":471859,"inputModalities":["text","image","audio"],"outputModalities":["text"],"price":{"inputPerMTok":1,"outputPerMTok":4.05,"cachedInputPerMTok":0.16999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling"},"blendedPerMTok":1.7625,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"thinkingmachines/Inkling","benchmarks":{"intelligence_index":25.5,"coding_index":52.1,"agentic_index":24.3},"releasedAt":"2026-07-17T22:05:56.000Z","isFree":true},{"id":"openrouter/auto-beta","slug":"openrouter-auto-beta","name":"Auto Router (Beta)","shortName":"Auto Router (Beta)","providerId":"openrouter","description":"Auto Router (Beta) is a task-aware router from OpenRouter. It classifies each request, then routes it the [most popular model](/rankings#task-spend) for that task based on aggregate spend, filtered by your...","contextLength":2000000,"maxOutputTokens":null,"inputModalities":["text","image","audio","file","video"],"outputModalities":["text","image"],"price":{"inputPerMTok":-1000000,"outputPerMTok":-1000000,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openrouter/auto-beta"},"blendedPerMTok":-1000000,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-07-17T17:59:25.000Z","isFree":false},{"id":"moonshotai/kimi-k3","slug":"moonshotai-kimi-k3","name":"MoonshotAI: Kimi K3","shortName":"Kimi K3","providerId":"moonshotai","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","contextLength":1048576,"maxOutputTokens":943718,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":2.6481380629999998,"outputPerMTok":13.282724250000001,"cachedInputPerMTok":0.30264435,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k3"},"blendedPerMTok":5.30678460975,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"moonshotai/Kimi-K3","benchmarks":{"intelligence_index":43.8,"coding_index":76.2,"agentic_index":50.6},"releasedAt":"2026-07-16T15:30:58.000Z","isFree":false},{"id":"meta/muse-spark-1.1","slug":"meta-muse-spark-1-1","name":"Meta: Muse Spark 1.1","shortName":"Muse Spark 1.1","providerId":"meta","description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...","contextLength":1048576,"maxOutputTokens":943718,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":4.25,"cachedInputPerMTok":0.15,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.0025,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.1"},"blendedPerMTok":2,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":34.3,"coding_index":71.3,"agentic_index":27.5},"releasedAt":"2026-07-16T15:29:01.000Z","isFree":false},{"id":"kwaipilot/kat-coder-pro-v2.5","slug":"kwaipilot-kat-coder-pro-v2-5","name":"Kwaipilot: KAT-Coder-Pro V2.5","shortName":"KAT-Coder-Pro V2.5","providerId":"kwaipilot","description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.74,"outputPerMTok":2.96,"cachedInputPerMTok":0.15,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/kwaipilot/kat-coder-pro-v2.5"},"blendedPerMTok":1.295,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-07-10T20:16:29.000Z","isFree":false},{"id":"openai/gpt-5.6-luna-pro","slug":"openai-gpt-5-6-luna-pro","name":"OpenAI: GPT-5.6 Luna Pro","shortName":"GPT-5.6 Luna Pro","providerId":"openai","description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":1.2,"cachedInputPerMTok":0.02,"cacheWritePerMTok":0.25,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-luna-pro"},"blendedPerMTok":0.44999999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2026-02-16","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-07-09T09:54:27.000Z","isFree":false},{"id":"openai/gpt-5.6-luna","slug":"openai-gpt-5-6-luna","name":"OpenAI: GPT-5.6 Luna","shortName":"GPT-5.6 Luna","providerId":"openai","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":1.2,"cachedInputPerMTok":0.02,"cacheWritePerMTok":0.25,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-luna"},"blendedPerMTok":0.44999999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2026-02-16","openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":37.5,"coding_index":71.4,"agentic_index":42.7},"releasedAt":"2026-07-09T09:54:24.000Z","isFree":false},{"id":"openai/gpt-5.6-terra-pro","slug":"openai-gpt-5-6-terra-pro","name":"OpenAI: GPT-5.6 Terra Pro","shortName":"GPT-5.6 Terra Pro","providerId":"openai","description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":12,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":2.5,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-terra-pro"},"blendedPerMTok":4.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2026-02-16","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-07-09T09:54:21.000Z","isFree":false},{"id":"openai/gpt-5.6-terra","slug":"openai-gpt-5-6-terra","name":"OpenAI: GPT-5.6 Terra","shortName":"GPT-5.6 Terra","providerId":"openai","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":12,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":2.5,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-terra"},"blendedPerMTok":4.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2026-02-16","openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":42.3,"coding_index":76.7,"agentic_index":43.7},"releasedAt":"2026-07-09T09:54:17.000Z","isFree":false},{"id":"openai/gpt-5.6-sol-pro","slug":"openai-gpt-5-6-sol-pro","name":"OpenAI: GPT-5.6 Sol Pro","shortName":"GPT-5.6 Sol Pro","providerId":"openai","description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":10,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":2.5,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-sol-pro"},"blendedPerMTok":4,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2026-02-16","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-07-09T09:54:14.000Z","isFree":false},{"id":"openai/gpt-5.6-sol","slug":"openai-gpt-5-6-sol","name":"OpenAI: GPT-5.6 Sol","shortName":"GPT-5.6 Sol","providerId":"openai","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":10,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":2.5,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-sol"},"blendedPerMTok":4,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2026-02-16","openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":47.1,"coding_index":77.4,"agentic_index":50.5},"releasedAt":"2026-07-09T09:54:10.000Z","isFree":false},{"id":"x-ai/grok-4.5","slug":"x-ai-grok-4-5","name":"SpaceXAI: Grok 4.5","shortName":"Grok 4.5","providerId":"x-ai","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","contextLength":500000,"maxOutputTokens":450000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":6,"cachedInputPerMTok":0.3,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.005,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/x-ai/grok-4.5"},"blendedPerMTok":3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":39.1,"coding_index":72.4,"agentic_index":42.1},"releasedAt":"2026-07-08T15:05:54.000Z","isFree":false},{"id":"aion-labs/aion-3.0-mini","slug":"aion-labs-aion-3-0-mini","name":"AionLabs: Aion-3.0-Mini","shortName":"Aion-3.0-Mini","providerId":"aion-labs","description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","contextLength":131072,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.7,"outputPerMTok":1.4,"cachedInputPerMTok":0.18,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/aion-labs/aion-3.0-mini"},"blendedPerMTok":0.8749999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-07-07T16:51:36.000Z","isFree":false},{"id":"aion-labs/aion-3.0","slug":"aion-labs-aion-3-0","name":"AionLabs: Aion-3.0","shortName":"Aion-3.0","providerId":"aion-labs","description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","contextLength":131072,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":3,"outputPerMTok":6,"cachedInputPerMTok":0.75,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/aion-labs/aion-3.0"},"blendedPerMTok":3.75,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-07-07T16:51:35.000Z","isFree":false},{"id":"tencent/hy3","slug":"tencent-hy3","name":"Tencent: Hy3","shortName":"Hy3","providerId":"tencent","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","contextLength":262144,"maxOutputTokens":128000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.13199999999999998,"outputPerMTok":0.5279999999999999,"cachedInputPerMTok":0.032999999999999995,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/tencent/hy3"},"blendedPerMTok":0.23099999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"tencent/Hy3","benchmarks":null,"releasedAt":"2026-07-06T13:20:48.000Z","isFree":false},{"id":"poolside/laguna-xs-2.1","slug":"poolside-laguna-xs-2-1","name":"Poolside: Laguna XS 2.1","shortName":"Laguna XS 2.1","providerId":"poolside","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","contextLength":262144,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.06,"outputPerMTok":0.12,"cachedInputPerMTok":0.03,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/poolside/laguna-xs-2.1"},"blendedPerMTok":0.075,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"poolside/Laguna-XS-2.1","benchmarks":null,"releasedAt":"2026-07-02T14:27:09.000Z","isFree":true},{"id":"anthropic/claude-sonnet-5","slug":"anthropic-claude-sonnet-5","name":"Anthropic: Claude Sonnet 5","shortName":"Claude Sonnet 5","providerId":"anthropic","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","contextLength":1000000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":10,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":2.5,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-sonnet-5"},"blendedPerMTok":4,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":38.4,"coding_index":71.5,"agentic_index":44.3},"releasedAt":"2026-06-30T18:11:23.000Z","isFree":false},{"id":"google/gemini-3.1-flash-lite-image","slug":"google-gemini-3-1-flash-lite-image","name":"Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)","shortName":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)","providerId":"google","description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","contextLength":65536,"maxOutputTokens":58982,"inputModalities":["image","text"],"outputModalities":["image","text"],"price":{"inputPerMTok":0.25,"outputPerMTok":1.5,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.1-flash-lite-image"},"blendedPerMTok":0.5625,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-01-01","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-06-30T16:33:45.000Z","isFree":false},{"id":"sakana/fugu-ultra","slug":"sakana-fugu-ultra","name":"Sakana: Fugu Ultra","shortName":"Fugu Ultra","providerId":"sakana","description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","contextLength":1000000,"maxOutputTokens":128000,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":5,"outputPerMTok":30,"cachedInputPerMTok":0.5,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/sakana/fugu-ultra"},"blendedPerMTok":11.25,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-06-24T04:45:03.000Z","isFree":false},{"id":"google/gemini-3.1-flash-image","slug":"google-gemini-3-1-flash-image","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image)","shortName":"Nano Banana 2 (Gemini 3.1 Flash Image)","providerId":"google","description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","contextLength":131072,"maxOutputTokens":32768,"inputModalities":["image","text"],"outputModalities":["image","text"],"price":{"inputPerMTok":0.5,"outputPerMTok":3,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.1-flash-image"},"blendedPerMTok":1.125,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-06-18T03:41:05.000Z","isFree":false},{"id":"google/gemini-3-pro-image","slug":"google-gemini-3-pro-image","name":"Google: Nano Banana Pro (Gemini 3 Pro Image)","shortName":"Nano Banana Pro (Gemini 3 Pro Image)","providerId":"google","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","contextLength":131072,"maxOutputTokens":32768,"inputModalities":["image","text"],"outputModalities":["image","text"],"price":{"inputPerMTok":2,"outputPerMTok":12,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":0.375,"perRequest":null,"perImage":0.000002,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3-pro-image"},"blendedPerMTok":4.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-06-18T03:40:54.000Z","isFree":false},{"id":"cohere/north-mini-code","slug":"cohere-north-mini-code","name":"Cohere: North Mini Code (free)","shortName":"North Mini Code (free)","providerId":"cohere","description":"North Mini Code is Cohere's first agentic coding model and the debut of its North family. A sparse mixture-of-experts model with 30B total parameters and 3B active, it is optimized...","contextLength":256000,"maxOutputTokens":64000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0,"outputPerMTok":0,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/cohere/north-mini-code"},"blendedPerMTok":0,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"CohereLabs/North-Mini-Code-1.0","benchmarks":{"coding_index":36.5,"agentic_index":1.1},"releasedAt":"2026-06-17T19:15:48.000Z","isFree":true},{"id":"z-ai/glm-5.2","slug":"z-ai-glm-5-2","name":"Z.ai: GLM 5.2","shortName":"GLM 5.2","providerId":"z-ai","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","contextLength":1048576,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.6832,"outputPerMTok":2.1471999999999998,"cachedInputPerMTok":0.12688000000000002,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-5.2"},"blendedPerMTok":1.0492,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"zai-org/GLM-5.2","benchmarks":{"coding_index":68.8,"agentic_index":39.4},"releasedAt":"2026-06-16T17:45:30.000Z","isFree":false},{"id":"openrouter/fusion","slug":"openrouter-fusion","name":"OpenRouter: Fusion","shortName":"Fusion","providerId":"openrouter","description":"Fusion turns your prompt into a small multi-model deliberation. A panel of expert models (see below) analyzes your prompt in parallel with web search and web fetch enabled, then a...","contextLength":1000000,"maxOutputTokens":null,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":-1000000,"outputPerMTok":-1000000,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openrouter/fusion"},"blendedPerMTok":-1000000,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-06-13T17:27:27.000Z","isFree":false},{"id":"moonshotai/kimi-k2.7-code","slug":"moonshotai-kimi-k2-7-code","name":"MoonshotAI: Kimi K2.7 Code","shortName":"Kimi K2.7 Code","providerId":"moonshotai","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.71,"outputPerMTok":3.5,"cachedInputPerMTok":0.15,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2.7-code"},"blendedPerMTok":1.4075,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"moonshotai/Kimi-K2.7-Code","benchmarks":{"intelligence_index":26.3,"coding_index":60.8,"agentic_index":22.5},"releasedAt":"2026-06-12T12:12:41.000Z","isFree":false},{"id":"anthropic/claude-fable-5","slug":"anthropic-claude-fable-5","name":"Anthropic: Claude Fable 5","shortName":"Claude Fable 5","providerId":"anthropic","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","contextLength":1000000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":10,"outputPerMTok":50,"cachedInputPerMTok":1,"cacheWritePerMTok":12.5,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-fable-5"},"blendedPerMTok":20,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":49.7,"coding_index":76.5,"agentic_index":51},"releasedAt":"2026-06-09T12:18:35.000Z","isFree":false},{"id":"nvidia/nemotron-3.5-content-safety","slug":"nvidia-nemotron-3-5-content-safety","name":"NVIDIA: Nemotron 3.5 Content Safety","shortName":"Nemotron 3.5 Content Safety","providerId":"nvidia","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":0.19999999999999998,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3.5-content-safety"},"blendedPerMTok":0.19999999999999998,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"nvidia/Nemotron-3.5-Content-Safety","benchmarks":null,"releasedAt":"2026-06-04T14:04:24.000Z","isFree":true},{"id":"nvidia/nemotron-3-ultra-550b-a55b","slug":"nvidia-nemotron-3-ultra-550b-a55b","name":"NVIDIA: Nemotron 3 Ultra","shortName":"Nemotron 3 Ultra","providerId":"nvidia","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","contextLength":262144,"maxOutputTokens":182520,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.6,"outputPerMTok":2.4,"cachedInputPerMTok":0.12,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b"},"blendedPerMTok":1.0499999999999998,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","benchmarks":{"intelligence_index":23.4,"coding_index":49.3,"agentic_index":21.7},"releasedAt":"2026-06-04T05:33:28.000Z","isFree":true},{"id":"qwen/qwen3.7-plus","slug":"qwen-qwen3-7-plus","name":"Qwen: Qwen3.7 Plus","shortName":"Qwen3.7 Plus","providerId":"qwen","description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","contextLength":1000000,"maxOutputTokens":131072,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.32,"outputPerMTok":1.28,"cachedInputPerMTok":0.064,"cacheWritePerMTok":0.39999999999999997,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-plus"},"blendedPerMTok":0.56,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":25.8,"coding_index":55.9,"agentic_index":19.7},"releasedAt":"2026-06-03T13:03:03.000Z","isFree":false},{"id":"minimax/minimax-m3","slug":"minimax-minimax-m3","name":"MiniMax: MiniMax M3","shortName":"MiniMax M3","providerId":"minimax","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","contextLength":1048576,"maxOutputTokens":512000,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":1.2,"cachedInputPerMTok":0.06,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/minimax/minimax-m3"},"blendedPerMTok":0.5249999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"MiniMaxAI/Minimax-M3","benchmarks":{"intelligence_index":29.6,"coding_index":58.6,"agentic_index":30.8},"releasedAt":"2026-05-31T16:36:14.000Z","isFree":false},{"id":"stepfun/step-3.7-flash","slug":"stepfun-step-3-7-flash","name":"StepFun: Step 3.7 Flash","shortName":"Step 3.7 Flash","providerId":"stepfun","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","contextLength":262144,"maxOutputTokens":230400,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":1.15,"cachedInputPerMTok":0.04,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/stepfun/step-3.7-flash"},"blendedPerMTok":0.4375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"stepfun-ai/Step-3.7-Flash","benchmarks":{"coding_index":39.6},"releasedAt":"2026-05-28T16:17:49.000Z","isFree":false},{"id":"anthropic/claude-opus-4.8","slug":"anthropic-claude-opus-4-8","name":"Anthropic: Claude Opus 4.8","shortName":"Claude Opus 4.8","providerId":"anthropic","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","contextLength":1000000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":5,"outputPerMTok":25,"cachedInputPerMTok":0.5,"cacheWritePerMTok":6.25,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.8"},"blendedPerMTok":10,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":42,"coding_index":74.3,"agentic_index":42.6},"releasedAt":"2026-05-27T18:04:51.000Z","isFree":false},{"id":"qwen/qwen3.7-max","slug":"qwen-qwen3-7-max","name":"Qwen: Qwen3.7 Max","shortName":"Qwen3.7 Max","providerId":"qwen","description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","contextLength":1000000,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":1.475,"outputPerMTok":4.425,"cachedInputPerMTok":0.295,"cacheWritePerMTok":1.84375,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-max"},"blendedPerMTok":2.2125000000000004,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":29.9,"coding_index":66,"agentic_index":23.9},"releasedAt":"2026-05-21T15:21:01.000Z","isFree":false},{"id":"x-ai/grok-build-0.1","slug":"x-ai-grok-build-0-1","name":"SpaceXAI: Grok Build 0.1","shortName":"Grok Build 0.1","providerId":"x-ai","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","contextLength":256000,"maxOutputTokens":230400,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":1,"outputPerMTok":2,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.005,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/x-ai/grok-build-0.1"},"blendedPerMTok":1.25,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"coding_index":51.5},"releasedAt":"2026-05-20T17:28:43.000Z","isFree":false},{"id":"google/gemini-3.5-flash","slug":"google-gemini-3-5-flash","name":"Google: Gemini 3.5 Flash","shortName":"Gemini 3.5 Flash","providerId":"google","description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":1.5,"outputPerMTok":9,"cachedInputPerMTok":0.15,"cacheWritePerMTok":0.0833333333333333,"perRequest":null,"perImage":0.0000015,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash"},"blendedPerMTok":3.375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2025-01-01","openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":33,"coding_index":70.1,"agentic_index":27.3},"releasedAt":"2026-05-19T12:30:00.000Z","isFree":false},{"id":"perceptron/perceptron-mk1","slug":"perceptron-perceptron-mk1","name":"Perceptron: Perceptron Mk1","shortName":"Perceptron Mk1","providerId":"perceptron","description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","contextLength":32768,"maxOutputTokens":8192,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":1.5,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/perceptron/perceptron-mk1"},"blendedPerMTok":0.4875,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-05-12T14:43:49.000Z","isFree":false},{"id":"google/gemini-3.1-flash-lite","slug":"google-gemini-3-1-flash-lite","name":"Google: Gemini 3.1 Flash Lite","shortName":"Gemini 3.1 Flash Lite","providerId":"google","description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":0.25,"outputPerMTok":1.5,"cachedInputPerMTok":0.024999999999999998,"cacheWritePerMTok":0.0833333333333333,"perRequest":null,"perImage":2.5e-7,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.1-flash-lite"},"blendedPerMTok":0.5625,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-05-07T15:47:08.000Z","isFree":false},{"id":"openai/gpt-chat-latest","slug":"openai-gpt-chat-latest","name":"OpenAI: GPT Chat Latest","shortName":"GPT Chat Latest","providerId":"openai","description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":5,"outputPerMTok":30,"cachedInputPerMTok":0.5,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-chat-latest"},"blendedPerMTok":11.25,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-05-05T16:56:52.000Z","isFree":false},{"id":"x-ai/grok-4.3","slug":"x-ai-grok-4-3","name":"SpaceXAI: Grok 4.3","shortName":"Grok 4.3","providerId":"x-ai","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","contextLength":1000000,"maxOutputTokens":900000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":2.5,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.005,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/x-ai/grok-4.3"},"blendedPerMTok":1.5625,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":25.4,"coding_index":42.2,"agentic_index":17.2},"releasedAt":"2026-04-30T23:30:21.000Z","isFree":false},{"id":"mistralai/mistral-medium-3-5","slug":"mistralai-mistral-medium-3-5","name":"Mistral: Mistral Medium 3.5","shortName":"Mistral Medium 3.5","providerId":"mistralai","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","contextLength":262144,"maxOutputTokens":209715,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":1.5,"outputPerMTok":7.5,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-medium-3-5"},"blendedPerMTok":3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":14.9,"coding_index":46.9,"agentic_index":9.4},"releasedAt":"2026-04-30T17:33:59.000Z","isFree":false},{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning","slug":"nvidia-nemotron-3-nano-omni-30b-a3b-reasoning","name":"NVIDIA: Nemotron 3 Nano Omni (free)","shortName":"Nemotron 3 Nano Omni (free)","providerId":"nvidia","description":"NVIDIA Nemotron™ 3 Nano Omni is a 30B-A3B open multimodal model designed to function as a perception and context sub-agent in enterprise agent systems. It accepts text, image, video, and...","contextLength":256000,"maxOutputTokens":65536,"inputModalities":["text","audio","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0,"outputPerMTok":0,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning"},"blendedPerMTok":0,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16","benchmarks":{"coding_index":13.8},"releasedAt":"2026-04-28T16:18:15.000Z","isFree":true},{"id":"qwen/qwen3.5-plus-20260420","slug":"qwen-qwen3-5-plus-20260420","name":"Qwen: Qwen3.5 Plus 2026-04-20","shortName":"Qwen3.5 Plus 2026-04-20","providerId":"qwen","description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","contextLength":1000000,"maxOutputTokens":65536,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":1.7999999999999998,"cachedInputPerMTok":null,"cacheWritePerMTok":0.375,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.5-plus-20260420"},"blendedPerMTok":0.6749999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-04-27T03:42:48.000Z","isFree":false},{"id":"qwen/qwen3.6-flash","slug":"qwen-qwen3-6-flash","name":"Qwen: Qwen3.6 Flash","shortName":"Qwen3.6 Flash","providerId":"qwen","description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","contextLength":1000000,"maxOutputTokens":65536,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.1875,"outputPerMTok":1.125,"cachedInputPerMTok":null,"cacheWritePerMTok":0.234375,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-flash"},"blendedPerMTok":0.421875,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-04-27T03:42:42.000Z","isFree":false},{"id":"qwen/qwen3.6-35b-a3b","slug":"qwen-qwen3-6-35b-a3b","name":"Qwen: Qwen3.6 35B A3B","shortName":"Qwen3.6 35B A3B","providerId":"qwen","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.8999999999999999,"cachedInputPerMTok":0.049999999999999996,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-35b-a3b"},"blendedPerMTok":0.3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3.6-35B-A3B","benchmarks":{"intelligence_index":18.8,"coding_index":41.9,"agentic_index":15},"releasedAt":"2026-04-27T03:24:15.000Z","isFree":false},{"id":"qwen/qwen3.6-max-preview","slug":"qwen-qwen3-6-max-preview","name":"Qwen: Qwen3.6 Max Preview","shortName":"Qwen3.6 Max Preview","providerId":"qwen","description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","contextLength":262144,"maxOutputTokens":65536,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":1.0270000000000001,"outputPerMTok":6.162,"cachedInputPerMTok":null,"cacheWritePerMTok":1.28375,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-max-preview"},"blendedPerMTok":2.31075,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-04-27T03:24:02.000Z","isFree":false},{"id":"qwen/qwen3.6-27b","slug":"qwen-qwen3-6-27b","name":"Qwen: Qwen3.6 27B","shortName":"Qwen3.6 27B","providerId":"qwen","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities - accepting text, image, and video inputs...","contextLength":262144,"maxOutputTokens":65536,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":2,"cachedInputPerMTok":0.03,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-27b"},"blendedPerMTok":0.725,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3.6-27B","benchmarks":{"intelligence_index":21.9,"coding_index":53.7,"agentic_index":20.1},"releasedAt":"2026-04-27T01:57:44.000Z","isFree":false},{"id":"openai/gpt-5.5-pro","slug":"openai-gpt-5-5-pro","name":"OpenAI: GPT-5.5 Pro","shortName":"GPT-5.5 Pro","providerId":"openai","description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":30,"outputPerMTok":180,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.5-pro"},"blendedPerMTok":67.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":true},"knowledgeCutoff":"2025-12-01","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-04-24T17:31:36.000Z","isFree":false},{"id":"openai/gpt-5.5","slug":"openai-gpt-5-5","name":"OpenAI: GPT-5.5","shortName":"GPT-5.5","providerId":"openai","description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":5,"outputPerMTok":30,"cachedInputPerMTok":0.5,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.5"},"blendedPerMTok":11.25,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2025-12-01","openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":38.6,"coding_index":74.9,"agentic_index":37.3},"releasedAt":"2026-04-24T17:31:33.000Z","isFree":false},{"id":"deepseek/deepseek-v4-pro","slug":"deepseek-deepseek-v4-pro","name":"DeepSeek: DeepSeek V4 Pro 0423","shortName":"DeepSeek V4 Pro 0423","providerId":"deepseek","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","contextLength":1048576,"maxOutputTokens":393216,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":1.5999999999999999,"outputPerMTok":3.1999999999999997,"cachedInputPerMTok":0.135,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-pro"},"blendedPerMTok":2,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V4-Pro","benchmarks":{"intelligence_index":30.9,"coding_index":59.4,"agentic_index":27.7},"releasedAt":"2026-04-24T03:17:59.000Z","isFree":false},{"id":"deepseek/deepseek-v4-flash","slug":"deepseek-deepseek-v4-flash","name":"DeepSeek: DeepSeek V4 Flash 0423","shortName":"DeepSeek V4 Flash 0423","providerId":"deepseek","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","contextLength":1048576,"maxOutputTokens":384000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.08553999999999999,"outputPerMTok":0.17107999999999998,"cachedInputPerMTok":0.017108,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-flash"},"blendedPerMTok":0.10692499999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V4-Flash","benchmarks":{"intelligence_index":24.8,"coding_index":52,"agentic_index":27.9},"releasedAt":"2026-04-24T03:17:46.000Z","isFree":false},{"id":"tencent/hy3-preview","slug":"tencent-hy3-preview","name":"Tencent: Hy3 preview","shortName":"Hy3 preview","providerId":"tencent","description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.18,"outputPerMTok":0.6,"cachedInputPerMTok":0.06,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/tencent/hy3-preview"},"blendedPerMTok":0.28500000000000003,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"tencent/Hy3-preview","benchmarks":{"intelligence_index":25.8,"coding_index":58.8,"agentic_index":25.6},"releasedAt":"2026-04-22T17:15:50.000Z","isFree":false},{"id":"xiaomi/mimo-v2.5-pro","slug":"xiaomi-mimo-v2-5-pro","name":"Xiaomi: MiMo-V2.5-Pro","shortName":"MiMo-V2.5-Pro","providerId":"xiaomi","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","contextLength":1050000,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.435,"outputPerMTok":0.87,"cachedInputPerMTok":0.0036,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/xiaomi/mimo-v2.5-pro"},"blendedPerMTok":0.54375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"XiaomiMiMo/MiMo-V2.5-Pro","benchmarks":{"intelligence_index":26.4,"coding_index":60.2,"agentic_index":22.7},"releasedAt":"2026-04-22T16:11:13.000Z","isFree":false},{"id":"xiaomi/mimo-v2.5","slug":"xiaomi-mimo-v2-5","name":"Xiaomi: MiMo-V2.5","shortName":"MiMo-V2.5","providerId":"xiaomi","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","contextLength":1050000,"maxOutputTokens":131072,"inputModalities":["text","audio","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.14,"outputPerMTok":0.28,"cachedInputPerMTok":0.0028,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/xiaomi/mimo-v2.5"},"blendedPerMTok":0.17500000000000002,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"XiaomiMiMo/MiMo-V2.5","benchmarks":{"intelligence_index":22.3,"coding_index":56.8,"agentic_index":17.4},"releasedAt":"2026-04-22T16:11:09.000Z","isFree":false},{"id":"openai/gpt-5.4-image-2","slug":"openai-gpt-5-4-image-2","name":"OpenAI: GPT-5.4 Image 2","shortName":"GPT-5.4 Image 2","providerId":"openai","description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","contextLength":272000,"maxOutputTokens":128000,"inputModalities":["image","text","file"],"outputModalities":["image","text"],"price":{"inputPerMTok":8,"outputPerMTok":15,"cachedInputPerMTok":2,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.4-image-2"},"blendedPerMTok":9.75,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-04-21T18:52:08.000Z","isFree":false},{"id":"openrouter/pareto-code","slug":"openrouter-pareto-code","name":"Pareto Code Router","shortName":"Pareto Code Router","providerId":"openrouter","description":"The Pareto Router maintains a tiered shortlist of strong coding models, ranked by [Artificial Analysis](https://artificialanalysis.ai/) coding percentiles. Set min_coding_score between 0 and 1 on the [pareto-router plugin](https://openrouter.ai/docs/guides/routing/routers/pareto-router#the-min_coding_score-parameter) to control how...","contextLength":2000000,"maxOutputTokens":null,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":-1000000,"outputPerMTok":-1000000,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openrouter/pareto-code"},"blendedPerMTok":-1000000,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-04-21T05:05:00.000Z","isFree":false},{"id":"moonshotai/kimi-k2.6","slug":"moonshotai-kimi-k2-6","name":"MoonshotAI: Kimi K2.6","shortName":"Kimi K2.6","providerId":"moonshotai","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.95,"outputPerMTok":4,"cachedInputPerMTok":0.16,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2.6"},"blendedPerMTok":1.7125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"moonshotai/Kimi-K2.6","benchmarks":{"coding_index":61.8,"agentic_index":22.1},"releasedAt":"2026-04-20T15:36:42.000Z","isFree":false},{"id":"anthropic/claude-opus-4.7","slug":"anthropic-claude-opus-4-7","name":"Anthropic: Claude Opus 4.7","shortName":"Claude Opus 4.7","providerId":"anthropic","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","contextLength":1000000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":5,"outputPerMTok":25,"cachedInputPerMTok":0.5,"cacheWritePerMTok":6.25,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.7"},"blendedPerMTok":10,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"coding_index":73.6,"agentic_index":39.5},"releasedAt":"2026-04-16T14:51:40.000Z","isFree":false},{"id":"z-ai/glm-5.1","slug":"z-ai-glm-5-1","name":"Z.ai: GLM 5.1","shortName":"GLM 5.1","providerId":"z-ai","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","contextLength":204800,"maxOutputTokens":128000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.966,"outputPerMTok":3.036,"cachedInputPerMTok":0.1794,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-5.1"},"blendedPerMTok":1.4834999999999998,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"zai-org/GLM-5.1","benchmarks":{"intelligence_index":26.4,"coding_index":55.8,"agentic_index":25.2},"releasedAt":"2026-04-07T16:07:05.000Z","isFree":false},{"id":"google/gemma-4-26b-a4b-it","slug":"google-gemma-4-26b-a4b-it","name":"Google: Gemma 4 26B A4B ","shortName":"Gemma 4 26B A4B ","providerId":"google","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference - delivering near-31B quality at...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["image","text","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.09,"outputPerMTok":0.3,"cachedInputPerMTok":0.049999999999999996,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemma-4-26b-a4b-it"},"blendedPerMTok":0.14250000000000002,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"google/gemma-4-26B-A4B-it","benchmarks":{"coding_index":39.3},"releasedAt":"2026-04-03T14:53:09.000Z","isFree":true},{"id":"google/gemma-4-31b-it","slug":"google-gemma-4-31b-it","name":"Google: Gemma 4 31B","shortName":"Gemma 4 31B","providerId":"google","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","contextLength":262144,"maxOutputTokens":16384,"inputModalities":["image","text","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.09,"outputPerMTok":0.33999999999999997,"cachedInputPerMTok":0.049999999999999996,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemma-4-31b-it"},"blendedPerMTok":0.1525,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"google/gemma-4-31B-it","benchmarks":{"intelligence_index":15.4,"coding_index":43.4,"agentic_index":6.7},"releasedAt":"2026-04-02T16:48:06.000Z","isFree":true},{"id":"qwen/qwen3.6-plus","slug":"qwen-qwen3-6-plus","name":"Qwen: Qwen3.6 Plus","shortName":"Qwen3.6 Plus","providerId":"qwen","description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","contextLength":1000000,"maxOutputTokens":65536,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.325,"outputPerMTok":1.95,"cachedInputPerMTok":null,"cacheWritePerMTok":0.40625,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-plus"},"blendedPerMTok":0.73125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":{"coding_index":54.5},"releasedAt":"2026-04-02T12:39:17.000Z","isFree":false},{"id":"z-ai/glm-5v-turbo","slug":"z-ai-glm-5v-turbo","name":"Z.ai: GLM 5V Turbo","shortName":"GLM 5V Turbo","providerId":"z-ai","description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","contextLength":202752,"maxOutputTokens":131072,"inputModalities":["image","text","video"],"outputModalities":["text"],"price":{"inputPerMTok":1.2,"outputPerMTok":4,"cachedInputPerMTok":0.24,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-5v-turbo"},"blendedPerMTok":1.9,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-04-01T16:37:38.000Z","isFree":false},{"id":"arcee-ai/trinity-large-thinking","slug":"arcee-ai-trinity-large-thinking","name":"Arcee AI: Trinity Large Thinking","shortName":"Trinity Large Thinking","providerId":"arcee-ai","description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","contextLength":262144,"maxOutputTokens":80000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.25,"outputPerMTok":0.7999999999999999,"cachedInputPerMTok":0.06,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/arcee-ai/trinity-large-thinking"},"blendedPerMTok":0.38749999999999996,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"arcee-ai/Trinity-Large-Thinking","benchmarks":{"intelligence_index":10.9,"coding_index":25.8,"agentic_index":1.2},"releasedAt":"2026-04-01T15:45:18.000Z","isFree":false},{"id":"x-ai/grok-4.20-multi-agent","slug":"x-ai-grok-4-20-multi-agent","name":"SpaceXAI: Grok 4.20 Multi-Agent","shortName":"Grok 4.20 Multi-Agent","providerId":"x-ai","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","contextLength":2000000,"maxOutputTokens":1800000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":2.5,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.005,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/x-ai/grok-4.20-multi-agent"},"blendedPerMTok":1.5625,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-09-01","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-03-31T17:45:58.000Z","isFree":false},{"id":"x-ai/grok-4.20","slug":"x-ai-grok-4-20","name":"SpaceXAI: Grok 4.20","shortName":"Grok 4.20","providerId":"x-ai","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","contextLength":2000000,"maxOutputTokens":1800000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":2.5,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.005,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/x-ai/grok-4.20"},"blendedPerMTok":1.5625,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-09-01","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-03-31T17:43:39.000Z","isFree":false},{"id":"google/lyria-3-pro-preview","slug":"google-lyria-3-pro-preview","name":"Google: Lyria 3 Pro Preview","shortName":"Lyria 3 Pro Preview","providerId":"google","description":"Full-length songs are priced at $0.08 per song. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate high-quality, 48kHz...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","image"],"outputModalities":["text","audio"],"price":{"inputPerMTok":0,"outputPerMTok":0,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/lyria-3-pro-preview"},"blendedPerMTok":0,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-03-30T21:48:06.000Z","isFree":true},{"id":"google/lyria-3-clip-preview","slug":"google-lyria-3-clip-preview","name":"Google: Lyria 3 Clip Preview","shortName":"Lyria 3 Clip Preview","providerId":"google","description":"30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","image"],"outputModalities":["text","audio"],"price":{"inputPerMTok":0,"outputPerMTok":0,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/lyria-3-clip-preview"},"blendedPerMTok":0,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-03-30T21:47:35.000Z","isFree":true},{"id":"kwaipilot/kat-coder-pro-v2","slug":"kwaipilot-kat-coder-pro-v2","name":"Kwaipilot: KAT-Coder-Pro V2","shortName":"KAT-Coder-Pro V2","providerId":"kwaipilot","description":"KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...","contextLength":262144,"maxOutputTokens":144000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":1.2,"cachedInputPerMTok":0.06,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/kwaipilot/kat-coder-pro-v2"},"blendedPerMTok":0.5249999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":{"coding_index":59.5},"releasedAt":"2026-03-27T22:08:30.000Z","isFree":false},{"id":"rekaai/reka-edge","slug":"rekaai-reka-edge","name":"Reka Edge","shortName":"Reka Edge","providerId":"rekaai","description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","contextLength":16384,"maxOutputTokens":14745,"inputModalities":["image","text","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.09999999999999999,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/rekaai/reka-edge"},"blendedPerMTok":0.09999999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"RekaAI/reka-edge-2603","benchmarks":null,"releasedAt":"2026-03-20T17:16:05.000Z","isFree":false},{"id":"minimax/minimax-m2.7","slug":"minimax-minimax-m2-7","name":"MiniMax: MiniMax M2.7","shortName":"MiniMax M2.7","providerId":"minimax","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","contextLength":204800,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":1.2,"cachedInputPerMTok":0.06,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/minimax/minimax-m2.7"},"blendedPerMTok":0.5249999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"MiniMaxAI/MiniMax-M2.7","benchmarks":{"intelligence_index":23.2,"coding_index":52.6,"agentic_index":16.8},"releasedAt":"2026-03-18T12:24:57.000Z","isFree":false},{"id":"openai/gpt-5.4-nano","slug":"openai-gpt-5-4-nano","name":"OpenAI: GPT-5.4 Nano","shortName":"GPT-5.4 Nano","providerId":"openai","description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":1.25,"cachedInputPerMTok":0.02,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.4-nano"},"blendedPerMTok":0.4625,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2025-08-31","openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":21.2,"coding_index":56.1,"agentic_index":17.7},"releasedAt":"2026-03-17T11:49:47.000Z","isFree":false},{"id":"openai/gpt-5.4-mini","slug":"openai-gpt-5-4-mini","name":"OpenAI: GPT-5.4 Mini","shortName":"GPT-5.4 Mini","providerId":"openai","description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":0.75,"outputPerMTok":4.5,"cachedInputPerMTok":0.075,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.4-mini"},"blendedPerMTok":1.6875,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2025-08-31","openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":24.6,"coding_index":56.1,"agentic_index":19.7},"releasedAt":"2026-03-17T11:49:38.000Z","isFree":false},{"id":"mistralai/mistral-small-2603","slug":"mistralai-mistral-small-2603","name":"Mistral: Mistral Small 4","shortName":"Mistral Small 4","providerId":"mistralai","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","contextLength":262144,"maxOutputTokens":209715,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":0.6,"cachedInputPerMTok":0.015,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-small-2603"},"blendedPerMTok":0.26249999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"mistralai/Mistral-Small-4-119B-2603","benchmarks":{"intelligence_index":11.5,"coding_index":26.6,"agentic_index":1.4},"releasedAt":"2026-03-16T21:14:45.000Z","isFree":false},{"id":"z-ai/glm-5-turbo","slug":"z-ai-glm-5-turbo","name":"Z.ai: GLM 5 Turbo","shortName":"GLM 5 Turbo","providerId":"z-ai","description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","contextLength":202752,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":1.2,"outputPerMTok":4,"cachedInputPerMTok":0.24,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-5-turbo"},"blendedPerMTok":1.9,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-03-15T14:06:13.000Z","isFree":false},{"id":"nvidia/nemotron-3-super-120b-a12b","slug":"nvidia-nemotron-3-super-120b-a12b","name":"NVIDIA: Nemotron 3 Super","shortName":"Nemotron 3 Super","providerId":"nvidia","description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.08,"outputPerMTok":0.44999999999999996,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3-super-120b-a12b"},"blendedPerMTok":0.1725,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","benchmarks":{"intelligence_index":13.6,"coding_index":37.7,"agentic_index":4.1},"releasedAt":"2026-03-11T16:07:19.000Z","isFree":true},{"id":"bytedance-seed/seed-2.0-lite","slug":"bytedance-seed-seed-2-0-lite","name":"ByteDance Seed: Seed-2.0-Lite","shortName":"Seed-2.0-Lite","providerId":"bytedance-seed","description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","contextLength":262144,"maxOutputTokens":131072,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.25,"outputPerMTok":2,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/bytedance-seed/seed-2.0-lite"},"blendedPerMTok":0.6875,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-03-10T15:40:31.000Z","isFree":false},{"id":"qwen/qwen3.5-9b","slug":"qwen-qwen3-5-9b","name":"Qwen: Qwen3.5-9B","shortName":"Qwen3.5-9B","providerId":"qwen","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.15,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.5-9b"},"blendedPerMTok":0.11249999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3.5-9B","benchmarks":{"coding_index":28.7},"releasedAt":"2026-03-10T14:19:56.000Z","isFree":false},{"id":"openai/gpt-5.4-pro","slug":"openai-gpt-5-4-pro","name":"OpenAI: GPT-5.4 Pro","shortName":"GPT-5.4 Pro","providerId":"openai","description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":30,"outputPerMTok":180,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.4-pro"},"blendedPerMTok":67.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-03-05T18:12:46.000Z","isFree":false},{"id":"openai/gpt-5.4","slug":"openai-gpt-5-4","name":"OpenAI: GPT-5.4","shortName":"GPT-5.4","providerId":"openai","description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","contextLength":1050000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":2.5,"outputPerMTok":15,"cachedInputPerMTok":0.25,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.4"},"blendedPerMTok":5.625,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":{"coding_index":71.1},"releasedAt":"2026-03-05T18:12:32.000Z","isFree":false},{"id":"inception/mercury-2","slug":"inception-mercury-2","name":"Inception: Mercury 2","shortName":"Mercury 2","providerId":"inception","description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","contextLength":128000,"maxOutputTokens":50000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.25,"outputPerMTok":0.75,"cachedInputPerMTok":0.024999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/inception/mercury-2"},"blendedPerMTok":0.375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":{"intelligence_index":11.5,"coding_index":31.1,"agentic_index":4},"releasedAt":"2026-03-04T14:57:55.000Z","isFree":false},{"id":"google/gemini-3.1-flash-lite-preview","slug":"google-gemini-3-1-flash-lite-preview","name":"Google: Gemini 3.1 Flash Lite Preview","shortName":"Gemini 3.1 Flash Lite Preview","providerId":"google","description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","image","video","file","audio"],"outputModalities":["text"],"price":{"inputPerMTok":0.25,"outputPerMTok":1.5,"cachedInputPerMTok":0.024999999999999998,"cacheWritePerMTok":0.0833333333333333,"perRequest":null,"perImage":2.5e-7,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.1-flash-lite-preview"},"blendedPerMTok":0.5625,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":16,"coding_index":34.7,"agentic_index":3.2},"releasedAt":"2026-03-03T04:37:53.000Z","isFree":false},{"id":"bytedance-seed/seed-2.0-mini","slug":"bytedance-seed-seed-2-0-mini","name":"ByteDance Seed: Seed-2.0-Mini","shortName":"Seed-2.0-Mini","providerId":"bytedance-seed","description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","contextLength":262144,"maxOutputTokens":131072,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.39999999999999997,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/bytedance-seed/seed-2.0-mini"},"blendedPerMTok":0.175,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-02-26T18:38:27.000Z","isFree":false},{"id":"google/gemini-3.1-flash-image-preview","slug":"google-gemini-3-1-flash-image-preview","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)","shortName":"Nano Banana 2 (Gemini 3.1 Flash Image Preview)","providerId":"google","description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","contextLength":65536,"maxOutputTokens":58982,"inputModalities":["image","text"],"outputModalities":["image","text"],"price":{"inputPerMTok":0.5,"outputPerMTok":3,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.1-flash-image-preview"},"blendedPerMTok":1.125,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-02-26T15:25:58.000Z","isFree":false},{"id":"qwen/qwen3.5-35b-a3b","slug":"qwen-qwen3-5-35b-a3b","name":"Qwen: Qwen3.5-35B-A3B","shortName":"Qwen3.5-35B-A3B","providerId":"qwen","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","contextLength":262144,"maxOutputTokens":16384,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.3125,"outputPerMTok":1.25,"cachedInputPerMTok":0.15625,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.5-35b-a3b"},"blendedPerMTok":0.546875,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3.5-35B-A3B","benchmarks":{"coding_index":37},"releasedAt":"2026-02-25T21:10:22.000Z","isFree":false},{"id":"qwen/qwen3.5-27b","slug":"qwen-qwen3-5-27b","name":"Qwen: Qwen3.5-27B","shortName":"Qwen3.5-27B","providerId":"qwen","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","contextLength":262144,"maxOutputTokens":65536,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.195,"outputPerMTok":1.56,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.5-27b"},"blendedPerMTok":0.53625,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3.5-27B","benchmarks":null,"releasedAt":"2026-02-25T21:10:10.000Z","isFree":false},{"id":"qwen/qwen3.5-122b-a10b","slug":"qwen-qwen3-5-122b-a10b","name":"Qwen: Qwen3.5-122B-A10B","shortName":"Qwen3.5-122B-A10B","providerId":"qwen","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","contextLength":262144,"maxOutputTokens":65536,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.26,"outputPerMTok":2.08,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.5-122b-a10b"},"blendedPerMTok":0.7150000000000001,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3.5-122B-A10B","benchmarks":{"intelligence_index":16.2,"coding_index":45.7,"agentic_index":9.6},"releasedAt":"2026-02-25T21:09:49.000Z","isFree":false},{"id":"qwen/qwen3.5-flash-02-23","slug":"qwen-qwen3-5-flash-02-23","name":"Qwen: Qwen3.5-Flash","shortName":"Qwen3.5-Flash","providerId":"qwen","description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","contextLength":1000000,"maxOutputTokens":65536,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.065,"outputPerMTok":0.26,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.5-flash-02-23"},"blendedPerMTok":0.11375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-02-25T21:09:36.000Z","isFree":false},{"id":"google/gemini-3.1-pro-preview-customtools","slug":"google-gemini-3-1-pro-preview-customtools","name":"Google: Gemini 3.1 Pro Preview Custom Tools","shortName":"Gemini 3.1 Pro Preview Custom Tools","providerId":"google","description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","audio","image","video","file"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":12,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":0.375,"perRequest":null,"perImage":0.000002,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.1-pro-preview-customtools"},"blendedPerMTok":4.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-02-25T18:58:43.000Z","isFree":false},{"id":"openai/gpt-5.3-codex","slug":"openai-gpt-5-3-codex","name":"OpenAI: GPT-5.3-Codex","shortName":"GPT-5.3-Codex","providerId":"openai","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":1.75,"outputPerMTok":14,"cachedInputPerMTok":0.175,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.3-codex"},"blendedPerMTok":4.8125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-02-24T18:52:44.000Z","isFree":false},{"id":"aion-labs/aion-2.0","slug":"aion-labs-aion-2-0","name":"AionLabs: Aion-2.0","shortName":"Aion-2.0","providerId":"aion-labs","description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","contextLength":131072,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.7999999999999999,"outputPerMTok":1.5999999999999999,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/aion-labs/aion-2.0"},"blendedPerMTok":1,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-02-23T21:15:06.000Z","isFree":false},{"id":"google/gemini-3.1-pro-preview","slug":"google-gemini-3-1-pro-preview","name":"Google: Gemini 3.1 Pro Preview","shortName":"Gemini 3.1 Pro Preview","providerId":"google","description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["audio","file","image","text","video"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":12,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":0.375,"perRequest":null,"perImage":0.000002,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3.1-pro-preview"},"blendedPerMTok":4.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":30.4,"coding_index":68.8,"agentic_index":10.3},"releasedAt":"2026-02-19T14:00:27.000Z","isFree":false},{"id":"anthropic/claude-sonnet-4.6","slug":"anthropic-claude-sonnet-4-6","name":"Anthropic: Claude Sonnet 4.6","shortName":"Claude Sonnet 4.6","providerId":"anthropic","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","contextLength":1000000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":3,"outputPerMTok":15,"cachedInputPerMTok":0.3,"cacheWritePerMTok":3.75,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-sonnet-4.6"},"blendedPerMTok":6,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":30.5,"coding_index":63,"agentic_index":33.1},"releasedAt":"2026-02-17T15:43:10.000Z","isFree":false},{"id":"qwen/qwen3.5-plus-02-15","slug":"qwen-qwen3-5-plus-02-15","name":"Qwen: Qwen3.5 Plus 2026-02-15","shortName":"Qwen3.5 Plus 2026-02-15","providerId":"qwen","description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","contextLength":1000000,"maxOutputTokens":65536,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.26,"outputPerMTok":1.56,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.5-plus-02-15"},"blendedPerMTok":0.585,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-02-16T08:10:16.000Z","isFree":false},{"id":"qwen/qwen3.5-397b-a17b","slug":"qwen-qwen3-5-397b-a17b","name":"Qwen: Qwen3.5 397B A17B","shortName":"Qwen3.5 397B A17B","providerId":"qwen","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text","image","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.55,"outputPerMTok":3.5,"cachedInputPerMTok":0.22499999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3.5-397b-a17b"},"blendedPerMTok":1.2875,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3.5-397B-A17B","benchmarks":{"intelligence_index":19.1,"coding_index":48.2,"agentic_index":10.6},"releasedAt":"2026-02-16T06:23:38.000Z","isFree":false},{"id":"minimax/minimax-m2.5","slug":"minimax-minimax-m2-5","name":"MiniMax: MiniMax M2.5","shortName":"MiniMax M2.5","providerId":"minimax","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","contextLength":204800,"maxOutputTokens":128000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.27,"outputPerMTok":1.08,"cachedInputPerMTok":0.027,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/minimax/minimax-m2.5"},"blendedPerMTok":0.47250000000000003,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"MiniMaxAI/MiniMax-M2.5","benchmarks":null,"releasedAt":"2026-02-12T15:01:42.000Z","isFree":false},{"id":"z-ai/glm-5","slug":"z-ai-glm-5","name":"Z.ai: GLM 5","shortName":"GLM 5","providerId":"z-ai","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","contextLength":204800,"maxOutputTokens":128000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.6,"outputPerMTok":1.92,"cachedInputPerMTok":0.12,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-5"},"blendedPerMTok":0.9299999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"zai-org/GLM-5","benchmarks":null,"releasedAt":"2026-02-11T16:59:42.000Z","isFree":false},{"id":"qwen/qwen3-max-thinking","slug":"qwen-qwen3-max-thinking","name":"Qwen: Qwen3 Max Thinking","shortName":"Qwen3 Max Thinking","providerId":"qwen","description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","contextLength":262144,"maxOutputTokens":65536,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.78,"outputPerMTok":3.9,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-max-thinking"},"blendedPerMTok":1.56,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2026-02-09T21:18:21.000Z","isFree":false},{"id":"anthropic/claude-opus-4.6","slug":"anthropic-claude-opus-4-6","name":"Anthropic: Claude Opus 4.6","shortName":"Claude Opus 4.6","providerId":"anthropic","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","contextLength":1000000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":5,"outputPerMTok":25,"cachedInputPerMTok":0.5,"cacheWritePerMTok":6.25,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.6"},"blendedPerMTok":10,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-02-04T15:30:50.000Z","isFree":false},{"id":"qwen/qwen3-coder-next","slug":"qwen-qwen3-coder-next","name":"Qwen: Qwen3 Coder Next","shortName":"Qwen3 Coder Next","providerId":"qwen","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.12,"outputPerMTok":0.7999999999999999,"cachedInputPerMTok":0.07,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-coder-next"},"blendedPerMTok":0.29,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3-Coder-Next","benchmarks":{"intelligence_index":10.1,"coding_index":36.2,"agentic_index":3.6},"releasedAt":"2026-02-04T00:15:01.000Z","isFree":false},{"id":"openrouter/free","slug":"openrouter-free","name":"Free Models Router","shortName":"Free Models Router","providerId":"openrouter","description":"The simplest way to get free inference. openrouter/free is a router that selects free models at random from the models available on OpenRouter. The router smartly filters for models that...","contextLength":200000,"maxOutputTokens":null,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0,"outputPerMTok":0,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openrouter/free"},"blendedPerMTok":0,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-02-01T03:43:47.000Z","isFree":true},{"id":"stepfun/step-3.5-flash","slug":"stepfun-step-3-5-flash","name":"StepFun: Step 3.5 Flash","shortName":"Step 3.5 Flash","providerId":"stepfun","description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","contextLength":262144,"maxOutputTokens":65536,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.3,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/stepfun/step-3.5-flash"},"blendedPerMTok":0.15,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"stepfun-ai/Step-3.5-Flash","benchmarks":null,"releasedAt":"2026-01-29T23:12:17.000Z","isFree":false},{"id":"moonshotai/kimi-k2.5","slug":"moonshotai-kimi-k2-5","name":"MoonshotAI: Kimi K2.5","shortName":"Kimi K2.5","providerId":"moonshotai","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.44999999999999996,"outputPerMTok":2.25,"cachedInputPerMTok":0.07,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2.5"},"blendedPerMTok":0.8999999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"moonshotai/Kimi-K2.5","benchmarks":{"coding_index":46.8},"releasedAt":"2026-01-27T04:11:16.000Z","isFree":false},{"id":"upstage/solar-pro-3","slug":"upstage-solar-pro-3","name":"Upstage: Solar Pro 3","shortName":"Solar Pro 3","providerId":"upstage","description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":0.6,"cachedInputPerMTok":0.015,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/upstage/solar-pro-3"},"blendedPerMTok":0.26249999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":7.8,"coding_index":16.2,"agentic_index":1.4},"releasedAt":"2026-01-27T02:33:20.000Z","isFree":false},{"id":"minimax/minimax-m2-her","slug":"minimax-minimax-m2-her","name":"MiniMax: MiniMax M2-her","shortName":"MiniMax M2-her","providerId":"minimax","description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","contextLength":65536,"maxOutputTokens":2048,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":1.2,"cachedInputPerMTok":0.03,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/minimax/minimax-m2-her"},"blendedPerMTok":0.5249999999999999,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-01-23T14:07:19.000Z","isFree":false},{"id":"writer/palmyra-x5","slug":"writer-palmyra-x5","name":"Writer: Palmyra X5","shortName":"Palmyra X5","providerId":"writer","description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","contextLength":1040000,"maxOutputTokens":8192,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.6,"outputPerMTok":6,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/writer/palmyra-x5"},"blendedPerMTok":1.95,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-01-21T13:57:03.000Z","isFree":false},{"id":"openai/gpt-audio","slug":"openai-gpt-audio","name":"OpenAI: GPT Audio","shortName":"GPT Audio","providerId":"openai","description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","contextLength":128000,"maxOutputTokens":16384,"inputModalities":["text","audio"],"outputModalities":["text","audio"],"price":{"inputPerMTok":2.5,"outputPerMTok":10,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-audio"},"blendedPerMTok":4.375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-01-19T22:42:49.000Z","isFree":false},{"id":"openai/gpt-audio-mini","slug":"openai-gpt-audio-mini","name":"OpenAI: GPT Audio Mini","shortName":"GPT Audio Mini","providerId":"openai","description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","contextLength":128000,"maxOutputTokens":16384,"inputModalities":["text","audio"],"outputModalities":["text","audio"],"price":{"inputPerMTok":0.6,"outputPerMTok":2.4,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-audio-mini"},"blendedPerMTok":1.0499999999999998,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-01-19T21:50:19.000Z","isFree":false},{"id":"z-ai/glm-4.7-flash","slug":"z-ai-glm-4-7-flash","name":"Z.ai: GLM 4.7 Flash","shortName":"GLM 4.7 Flash","providerId":"z-ai","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","contextLength":200000,"maxOutputTokens":117964,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.060500000000000005,"outputPerMTok":0.39999999999999997,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-4.7-flash"},"blendedPerMTok":0.145375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"zai-org/GLM-4.7-Flash","benchmarks":null,"releasedAt":"2026-01-19T14:45:13.000Z","isFree":false},{"id":"openai/gpt-5.2-codex","slug":"openai-gpt-5-2-codex","name":"OpenAI: GPT-5.2-Codex","shortName":"GPT-5.2-Codex","providerId":"openai","description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":1.75,"outputPerMTok":14,"cachedInputPerMTok":0.175,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.2-codex"},"blendedPerMTok":4.8125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2026-01-14T16:48:35.000Z","isFree":false},{"id":"bytedance-seed/seed-1.6-flash","slug":"bytedance-seed-seed-1-6-flash","name":"ByteDance Seed: Seed 1.6 Flash","shortName":"Seed 1.6 Flash","providerId":"bytedance-seed","description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","contextLength":262144,"maxOutputTokens":32768,"inputModalities":["image","text","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.075,"outputPerMTok":0.3,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/bytedance-seed/seed-1.6-flash"},"blendedPerMTok":0.13124999999999998,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-12-23T15:50:11.000Z","isFree":false},{"id":"bytedance-seed/seed-1.6","slug":"bytedance-seed-seed-1-6","name":"ByteDance Seed: Seed 1.6","shortName":"Seed 1.6","providerId":"bytedance-seed","description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","contextLength":262144,"maxOutputTokens":32768,"inputModalities":["image","text","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.25,"outputPerMTok":2,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/bytedance-seed/seed-1.6"},"blendedPerMTok":0.6875,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-12-23T15:49:57.000Z","isFree":false},{"id":"minimax/minimax-m2.1","slug":"minimax-minimax-m2-1","name":"MiniMax: MiniMax M2.1","shortName":"MiniMax M2.1","providerId":"minimax","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","contextLength":204800,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":1.2,"cachedInputPerMTok":0.03,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/minimax/minimax-m2.1"},"blendedPerMTok":0.5249999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"MiniMaxAI/MiniMax-M2.1","benchmarks":null,"releasedAt":"2025-12-23T01:56:37.000Z","isFree":false},{"id":"z-ai/glm-4.7","slug":"z-ai-glm-4-7","name":"Z.ai: GLM 4.7","shortName":"GLM 4.7","providerId":"z-ai","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","contextLength":204800,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.39999999999999997,"outputPerMTok":1.75,"cachedInputPerMTok":0.08,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-4.7"},"blendedPerMTok":0.7375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"zai-org/GLM-4.7","benchmarks":{"coding_index":45.3},"releasedAt":"2025-12-22T04:33:34.000Z","isFree":false},{"id":"google/gemini-3-flash-preview","slug":"google-gemini-3-flash-preview","name":"Google: Gemini 3 Flash Preview","shortName":"Gemini 3 Flash Preview","providerId":"google","description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","image","file","audio","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.5,"outputPerMTok":3,"cachedInputPerMTok":0.049999999999999996,"cacheWritePerMTok":0.0833333333333333,"perRequest":null,"perImage":5e-7,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3-flash-preview"},"blendedPerMTok":1.125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-12-17T15:57:58.000Z","isFree":false},{"id":"nvidia/nemotron-3-nano-30b-a3b","slug":"nvidia-nemotron-3-nano-30b-a3b","name":"NVIDIA: Nemotron 3 Nano 30B A3B","shortName":"Nemotron 3 Nano 30B A3B","providerId":"nvidia","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.049999999999999996,"outputPerMTok":0.19999999999999998,"cachedInputPerMTok":0.03,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3-nano-30b-a3b"},"blendedPerMTok":0.0875,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","benchmarks":{"intelligence_index":8.9,"coding_index":14.4,"agentic_index":1},"releasedAt":"2025-12-14T16:54:35.000Z","isFree":false},{"id":"openai/gpt-5.2-chat","slug":"openai-gpt-5-2-chat","name":"OpenAI: GPT-5.2 Chat","shortName":"GPT-5.2 Chat","providerId":"openai","description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","contextLength":128000,"maxOutputTokens":32000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":1.75,"outputPerMTok":14,"cachedInputPerMTok":0.175,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.2-chat"},"blendedPerMTok":4.8125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-12-10T18:03:03.000Z","isFree":false},{"id":"openai/gpt-5.2-pro","slug":"openai-gpt-5-2-pro","name":"OpenAI: GPT-5.2 Pro","shortName":"GPT-5.2 Pro","providerId":"openai","description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":21,"outputPerMTok":168,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.2-pro"},"blendedPerMTok":57.75,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-12-10T18:03:00.000Z","isFree":false},{"id":"openai/gpt-5.2","slug":"openai-gpt-5-2","name":"OpenAI: GPT-5.2","shortName":"GPT-5.2","providerId":"openai","description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":1.75,"outputPerMTok":14,"cachedInputPerMTok":0.175,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.2"},"blendedPerMTok":4.8125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-12-10T18:02:55.000Z","isFree":false},{"id":"mistralai/devstral-2512","slug":"mistralai-devstral-2512","name":"Mistral: Devstral 2 2512","shortName":"Devstral 2 2512","providerId":"mistralai","description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","contextLength":262144,"maxOutputTokens":209715,"inputModalities":["text","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.39999999999999997,"outputPerMTok":2,"cachedInputPerMTok":0.04,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/devstral-2512"},"blendedPerMTok":0.8,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"mistralai/Devstral-2-123B-Instruct-2512","benchmarks":{"intelligence_index":9.4,"coding_index":31.3,"agentic_index":4.9},"releasedAt":"2025-12-09T13:03:39.000Z","isFree":false},{"id":"relace/relace-search","slug":"relace-relace-search","name":"Relace: Relace Search","shortName":"Relace Search","providerId":"relace","description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","contextLength":256000,"maxOutputTokens":128000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":1,"outputPerMTok":3,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/relace/relace-search"},"blendedPerMTok":1.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2025-12-08T17:06:00.000Z","isFree":false},{"id":"z-ai/glm-4.6v","slug":"z-ai-glm-4-6v","name":"Z.ai: GLM 4.6V","shortName":"GLM 4.6V","providerId":"z-ai","description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","contextLength":131072,"maxOutputTokens":32768,"inputModalities":["image","text","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":0.8999999999999999,"cachedInputPerMTok":0.055,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-4.6v"},"blendedPerMTok":0.44999999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"zai-org/GLM-4.6V","benchmarks":null,"releasedAt":"2025-12-08T15:24:22.000Z","isFree":false},{"id":"openrouter/bodybuilder","slug":"openrouter-bodybuilder","name":"Body Builder (beta)","shortName":"Body Builder (beta)","providerId":"openrouter","description":"Transform your natural language requests into structured OpenRouter API request objects. Describe what you want to accomplish with AI models, and Body Builder will construct the appropriate API calls. Example:...","contextLength":128000,"maxOutputTokens":null,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":-1000000,"outputPerMTok":-1000000,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openrouter/bodybuilder"},"blendedPerMTok":-1000000,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-12-05T03:00:53.000Z","isFree":false},{"id":"openai/gpt-5.1-codex-max","slug":"openai-gpt-5-1-codex-max","name":"OpenAI: GPT-5.1-Codex-Max","shortName":"GPT-5.1-Codex-Max","providerId":"openai","description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":10,"cachedInputPerMTok":0.125,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.1-codex-max"},"blendedPerMTok":3.4375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-12-04T20:08:54.000Z","isFree":false},{"id":"amazon/nova-2-lite-v1","slug":"amazon-nova-2-lite-v1","name":"Amazon: Nova 2 Lite","shortName":"Nova 2 Lite","providerId":"amazon","description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","contextLength":1000000,"maxOutputTokens":65535,"inputModalities":["text","image","video","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":2.5,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/amazon/nova-2-lite-v1"},"blendedPerMTok":0.85,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":{"coding_index":23},"releasedAt":"2025-12-02T17:31:12.000Z","isFree":false},{"id":"mistralai/ministral-14b-2512","slug":"mistralai-ministral-14b-2512","name":"Mistral: Ministral 3 14B 2512","shortName":"Ministral 3 14B 2512","providerId":"mistralai","description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","contextLength":262144,"maxOutputTokens":209715,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":0.19999999999999998,"cachedInputPerMTok":0.02,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/ministral-14b-2512"},"blendedPerMTok":0.19999999999999998,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"mistralai/Ministral-3-14B-Instruct-2512","benchmarks":{"intelligence_index":6,"coding_index":14.4,"agentic_index":1.1},"releasedAt":"2025-12-02T13:22:15.000Z","isFree":false},{"id":"mistralai/ministral-8b-2512","slug":"mistralai-ministral-8b-2512","name":"Mistral: Ministral 3 8B 2512","shortName":"Ministral 3 8B 2512","providerId":"mistralai","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","contextLength":262144,"maxOutputTokens":209715,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":0.15,"cachedInputPerMTok":0.015,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/ministral-8b-2512"},"blendedPerMTok":0.15,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"mistralai/Ministral-3-8B-Instruct-2512","benchmarks":{"intelligence_index":5.5,"coding_index":9.7,"agentic_index":0.6},"releasedAt":"2025-12-02T13:20:54.000Z","isFree":false},{"id":"mistralai/ministral-3b-2512","slug":"mistralai-ministral-3b-2512","name":"Mistral: Ministral 3 3B 2512","shortName":"Ministral 3 3B 2512","providerId":"mistralai","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","contextLength":131072,"maxOutputTokens":104857,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.09999999999999999,"cachedInputPerMTok":0.01,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/ministral-3b-2512"},"blendedPerMTok":0.09999999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"mistralai/Ministral-3-3B-Instruct-2512","benchmarks":{"intelligence_index":4.8,"coding_index":4.8,"agentic_index":0.8},"releasedAt":"2025-12-02T13:19:20.000Z","isFree":false},{"id":"mistralai/mistral-large-2512","slug":"mistralai-mistral-large-2512","name":"Mistral: Mistral Large 3 2512","shortName":"Mistral Large 3 2512","providerId":"mistralai","description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","contextLength":262144,"maxOutputTokens":209715,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.5,"outputPerMTok":1.5,"cachedInputPerMTok":0.049999999999999996,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-large-2512"},"blendedPerMTok":0.75,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":9.7,"coding_index":20.1,"agentic_index":2.4},"releasedAt":"2025-12-01T21:27:52.000Z","isFree":false},{"id":"deepseek/deepseek-v3.2","slug":"deepseek-deepseek-v3-2","name":"DeepSeek: DeepSeek V3.2","shortName":"DeepSeek V3.2","providerId":"deepseek","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","contextLength":163840,"maxOutputTokens":65536,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.26899999999999996,"outputPerMTok":0.39999999999999997,"cachedInputPerMTok":0.13449999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v3.2"},"blendedPerMTok":0.30174999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V3.2","benchmarks":{"coding_index":44.2},"releasedAt":"2025-12-01T13:10:42.000Z","isFree":false},{"id":"anthropic/claude-opus-4.5","slug":"anthropic-claude-opus-4-5","name":"Anthropic: Claude Opus 4.5","shortName":"Claude Opus 4.5","providerId":"anthropic","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","contextLength":200000,"maxOutputTokens":64000,"inputModalities":["file","image","text"],"outputModalities":["text"],"price":{"inputPerMTok":5,"outputPerMTok":25,"cachedInputPerMTok":0.5,"cacheWritePerMTok":6.25,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.5"},"blendedPerMTok":10,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-11-24T18:56:20.000Z","isFree":false},{"id":"google/gemini-3-pro-image-preview","slug":"google-gemini-3-pro-image-preview","name":"Google: Nano Banana Pro (Gemini 3 Pro Image Preview)","shortName":"Nano Banana Pro (Gemini 3 Pro Image Preview)","providerId":"google","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","contextLength":65536,"maxOutputTokens":32768,"inputModalities":["image","text"],"outputModalities":["image","text"],"price":{"inputPerMTok":2,"outputPerMTok":12,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":0.375,"perRequest":null,"perImage":0.000002,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-3-pro-image-preview"},"blendedPerMTok":4.5,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-11-20T15:49:57.000Z","isFree":false},{"id":"openai/gpt-5.1","slug":"openai-gpt-5-1","name":"OpenAI: GPT-5.1","shortName":"GPT-5.1","providerId":"openai","description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":10,"cachedInputPerMTok":0.125,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.1"},"blendedPerMTok":3.4375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":{"coding_index":49.4},"releasedAt":"2025-11-13T18:58:25.000Z","isFree":false},{"id":"openai/gpt-5.1-codex","slug":"openai-gpt-5-1-codex","name":"OpenAI: GPT-5.1-Codex","shortName":"GPT-5.1-Codex","providerId":"openai","description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":10,"cachedInputPerMTok":0.13,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.1-codex"},"blendedPerMTok":3.4375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-11-13T18:58:18.000Z","isFree":false},{"id":"openai/gpt-5.1-codex-mini","slug":"openai-gpt-5-1-codex-mini","name":"OpenAI: GPT-5.1-Codex-Mini","shortName":"GPT-5.1-Codex-Mini","providerId":"openai","description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["image","text"],"outputModalities":["text"],"price":{"inputPerMTok":0.25,"outputPerMTok":2,"cachedInputPerMTok":0.03,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5.1-codex-mini"},"blendedPerMTok":0.6875,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-11-13T18:17:00.000Z","isFree":false},{"id":"moonshotai/kimi-k2-thinking","slug":"moonshotai-kimi-k2-thinking","name":"MoonshotAI: Kimi K2 Thinking","shortName":"Kimi K2 Thinking","providerId":"moonshotai","description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","contextLength":262144,"maxOutputTokens":98304,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.6,"outputPerMTok":2.5,"cachedInputPerMTok":0.15,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2-thinking"},"blendedPerMTok":1.075,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"moonshotai/Kimi-K2-Thinking","benchmarks":{"coding_index":21},"releasedAt":"2025-11-06T14:50:22.000Z","isFree":false},{"id":"amazon/nova-premier-v1","slug":"amazon-nova-premier-v1","name":"Amazon: Nova Premier 1.0","shortName":"Nova Premier 1.0","providerId":"amazon","description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","contextLength":1000000,"maxOutputTokens":32000,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":2.5,"outputPerMTok":12.5,"cachedInputPerMTok":0.625,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/amazon/nova-premier-v1"},"blendedPerMTok":5,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-10-31T22:38:52.000Z","isFree":false},{"id":"perplexity/sonar-pro-search","slug":"perplexity-sonar-pro-search","name":"Perplexity: Sonar Pro Search","shortName":"Sonar Pro Search","providerId":"perplexity","description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","contextLength":200000,"maxOutputTokens":8000,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":3,"outputPerMTok":15,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.018,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/perplexity/sonar-pro-search"},"blendedPerMTok":6,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-10-30T19:59:26.000Z","isFree":false},{"id":"mistralai/voxtral-small-24b-2507","slug":"mistralai-voxtral-small-24b-2507","name":"Mistral: Voxtral Small 24B 2507","shortName":"Voxtral Small 24B 2507","providerId":"mistralai","description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","contextLength":32768,"maxOutputTokens":26214,"inputModalities":["text","audio","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.3,"cachedInputPerMTok":0.01,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/voxtral-small-24b-2507"},"blendedPerMTok":0.15,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"mistralai/Voxtral-Small-24B-2507","benchmarks":null,"releasedAt":"2025-10-30T14:39:04.000Z","isFree":false},{"id":"openai/gpt-oss-safeguard-20b","slug":"openai-gpt-oss-safeguard-20b","name":"OpenAI: gpt-oss-safeguard-20b","shortName":"gpt-oss-safeguard-20b","providerId":"openai","description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","contextLength":131072,"maxOutputTokens":65536,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.075,"outputPerMTok":0.3,"cachedInputPerMTok":0.0375,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-oss-safeguard-20b"},"blendedPerMTok":0.13124999999999998,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"openai/gpt-oss-safeguard-20b","benchmarks":null,"releasedAt":"2025-10-29T15:47:16.000Z","isFree":false},{"id":"minimax/minimax-m2","slug":"minimax-minimax-m2","name":"MiniMax: MiniMax M2","shortName":"MiniMax M2","providerId":"minimax","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","contextLength":204800,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.255,"outputPerMTok":1.02,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/minimax/minimax-m2"},"blendedPerMTok":0.44625000000000004,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"MiniMaxAI/MiniMax-M2","benchmarks":null,"releasedAt":"2025-10-23T20:41:33.000Z","isFree":false},{"id":"qwen/qwen3-vl-32b-instruct","slug":"qwen-qwen3-vl-32b-instruct","name":"Qwen: Qwen3 VL 32B Instruct","shortName":"Qwen3 VL 32B Instruct","providerId":"qwen","description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","contextLength":131072,"maxOutputTokens":32768,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.10400000000000001,"outputPerMTok":0.41600000000000004,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-vl-32b-instruct"},"blendedPerMTok":0.18200000000000002,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3-VL-32B-Instruct","benchmarks":null,"releasedAt":"2025-10-23T14:55:32.000Z","isFree":false},{"id":"ibm-granite/granite-4.0-h-micro","slug":"ibm-granite-granite-4-0-h-micro","name":"IBM: Granite 4.0 Micro","shortName":"Granite 4.0 Micro","providerId":"ibm-granite","description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","contextLength":131000,"maxOutputTokens":117900,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.017,"outputPerMTok":0.112,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/ibm-granite/granite-4.0-h-micro"},"blendedPerMTok":0.04075,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"ibm-granite/granite-4.0-h-micro","benchmarks":null,"releasedAt":"2025-10-20T02:34:55.000Z","isFree":false},{"id":"openai/gpt-5-image-mini","slug":"openai-gpt-5-image-mini","name":"OpenAI: GPT-5 Image Mini","shortName":"GPT-5 Image Mini","providerId":"openai","description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["file","image","text"],"outputModalities":["image","text"],"price":{"inputPerMTok":2.5,"outputPerMTok":2,"cachedInputPerMTok":0.25,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5-image-mini"},"blendedPerMTok":2.375,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-10-16T14:23:03.000Z","isFree":false},{"id":"anthropic/claude-haiku-4.5","slug":"anthropic-claude-haiku-4-5","name":"Anthropic: Claude Haiku 4.5","shortName":"Claude Haiku 4.5","providerId":"anthropic","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","contextLength":200000,"maxOutputTokens":64000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":1,"outputPerMTok":5,"cachedInputPerMTok":0.09999999999999999,"cacheWritePerMTok":1.25,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-haiku-4.5"},"blendedPerMTok":2,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":17.6,"coding_index":43.9,"agentic_index":10.3},"releasedAt":"2025-10-15T17:00:38.000Z","isFree":false},{"id":"qwen/qwen3-vl-8b-thinking","slug":"qwen-qwen3-vl-8b-thinking","name":"Qwen: Qwen3 VL 8B Thinking","shortName":"Qwen3 VL 8B Thinking","providerId":"qwen","description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","contextLength":131072,"maxOutputTokens":32768,"inputModalities":["image","text"],"outputModalities":["text"],"price":{"inputPerMTok":0.18,"outputPerMTok":2.0999999999999996,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-vl-8b-thinking"},"blendedPerMTok":0.6599999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3-VL-8B-Thinking","benchmarks":null,"releasedAt":"2025-10-14T17:42:26.000Z","isFree":false},{"id":"qwen/qwen3-vl-8b-instruct","slug":"qwen-qwen3-vl-8b-instruct","name":"Qwen: Qwen3 VL 8B Instruct","shortName":"Qwen3 VL 8B Instruct","providerId":"qwen","description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","contextLength":262144,"maxOutputTokens":32768,"inputModalities":["image","text"],"outputModalities":["text"],"price":{"inputPerMTok":0.117,"outputPerMTok":0.45499999999999996,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-vl-8b-instruct"},"blendedPerMTok":0.2015,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":true,"huggingFaceId":"Qwen/Qwen3-VL-8B-Instruct","benchmarks":null,"releasedAt":"2025-10-14T17:35:08.000Z","isFree":false},{"id":"openai/gpt-5-image","slug":"openai-gpt-5-image","name":"OpenAI: GPT-5 Image","shortName":"GPT-5 Image","providerId":"openai","description":"[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["image","text","file"],"outputModalities":["image","text"],"price":{"inputPerMTok":10,"outputPerMTok":10,"cachedInputPerMTok":1.25,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5-image"},"blendedPerMTok":10,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-10-14T13:19:46.000Z","isFree":false},{"id":"google/gemini-2.5-flash-image","slug":"google-gemini-2-5-flash-image","name":"Google: Nano Banana (Gemini 2.5 Flash Image)","shortName":"Nano Banana (Gemini 2.5 Flash Image)","providerId":"google","description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","contextLength":32768,"maxOutputTokens":8192,"inputModalities":["image","text"],"outputModalities":["image","text"],"price":{"inputPerMTok":0.3,"outputPerMTok":2.5,"cachedInputPerMTok":0.03,"cacheWritePerMTok":0.0833333333333333,"perRequest":null,"perImage":3e-7,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-2.5-flash-image"},"blendedPerMTok":0.85,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-01-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-10-07T20:53:51.000Z","isFree":false},{"id":"qwen/qwen3-vl-30b-a3b-thinking","slug":"qwen-qwen3-vl-30b-a3b-thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","shortName":"Qwen3 VL 30B A3B Thinking","providerId":"qwen","description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","contextLength":262144,"maxOutputTokens":32768,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":2.4,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-vl-30b-a3b-thinking"},"blendedPerMTok":0.75,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"Qwen/Qwen3-VL-30B-A3B-Thinking","benchmarks":null,"releasedAt":"2025-10-06T23:47:59.000Z","isFree":false},{"id":"qwen/qwen3-vl-30b-a3b-instruct","slug":"qwen-qwen3-vl-30b-a3b-instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","shortName":"Qwen3 VL 30B A3B Instruct","providerId":"qwen","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","contextLength":262144,"maxOutputTokens":16384,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":0.6,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-vl-30b-a3b-instruct"},"blendedPerMTok":0.26249999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"Qwen/Qwen3-VL-30B-A3B-Instruct","benchmarks":null,"releasedAt":"2025-10-06T23:47:56.000Z","isFree":false},{"id":"openai/gpt-5-pro","slug":"openai-gpt-5-pro","name":"OpenAI: GPT-5 Pro","shortName":"GPT-5 Pro","providerId":"openai","description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":15,"outputPerMTok":120,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5-pro"},"blendedPerMTok":41.25,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":true},"knowledgeCutoff":"2024-09-30","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-10-06T18:51:03.000Z","isFree":false},{"id":"z-ai/glm-4.6","slug":"z-ai-glm-4-6","name":"Z.ai: GLM 4.6","shortName":"GLM 4.6","providerId":"z-ai","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","contextLength":204800,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.43,"outputPerMTok":1.75,"cachedInputPerMTok":0.08,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-4.6"},"blendedPerMTok":0.76,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"zai-org/GLM-4.6","benchmarks":{"coding_index":45.8},"releasedAt":"2025-09-30T12:32:56.000Z","isFree":false},{"id":"anthropic/claude-sonnet-4.5","slug":"anthropic-claude-sonnet-4-5","name":"Anthropic: Claude Sonnet 4.5","shortName":"Claude Sonnet 4.5","providerId":"anthropic","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","contextLength":1000000,"maxOutputTokens":64000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":3,"outputPerMTok":15,"cachedInputPerMTok":0.3,"cacheWritePerMTok":3.75,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-sonnet-4.5"},"blendedPerMTok":6,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2025-01-31","openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":21.2,"coding_index":52.1,"agentic_index":17.5},"releasedAt":"2025-09-29T16:01:16.000Z","isFree":false},{"id":"deepseek/deepseek-v3.2-exp","slug":"deepseek-deepseek-v3-2-exp","name":"DeepSeek: DeepSeek V3.2 Exp","shortName":"DeepSeek V3.2 Exp","providerId":"deepseek","description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","contextLength":163840,"maxOutputTokens":65536,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.27,"outputPerMTok":0.41,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v3.2-exp"},"blendedPerMTok":0.305,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-07-31","openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V3.2-Exp","benchmarks":null,"releasedAt":"2025-09-29T12:54:41.000Z","isFree":false},{"id":"thedrummer/cydonia-24b-v4.1","slug":"thedrummer-cydonia-24b-v4-1","name":"TheDrummer: Cydonia 24B V4.1","shortName":"Cydonia 24B V4.1","providerId":"thedrummer","description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":0.5,"cachedInputPerMTok":0.15,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/thedrummer/cydonia-24b-v4.1"},"blendedPerMTok":0.35,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-04-30","openWeights":true,"huggingFaceId":"thedrummer/cydonia-24b-v4.1","benchmarks":null,"releasedAt":"2025-09-27T00:11:18.000Z","isFree":false},{"id":"relace/relace-apply-3","slug":"relace-relace-apply-3","name":"Relace: Relace Apply 3","shortName":"Relace Apply 3","providerId":"relace","description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","contextLength":256000,"maxOutputTokens":128000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.85,"outputPerMTok":1.25,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/relace/relace-apply-3"},"blendedPerMTok":0.95,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-09-26T12:59:32.000Z","isFree":false},{"id":"qwen/qwen3-vl-235b-a22b-thinking","slug":"qwen-qwen3-vl-235b-a22b-thinking","name":"Qwen: Qwen3 VL 235B A22B Thinking","shortName":"Qwen3 VL 235B A22B Thinking","providerId":"qwen","description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","contextLength":131072,"maxOutputTokens":32768,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.39999999999999997,"outputPerMTok":4,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-vl-235b-a22b-thinking"},"blendedPerMTok":1.3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"Qwen/Qwen3-VL-235B-A22B-Thinking","benchmarks":null,"releasedAt":"2025-09-23T23:04:50.000Z","isFree":false},{"id":"qwen/qwen3-vl-235b-a22b-instruct","slug":"qwen-qwen3-vl-235b-a22b-instruct","name":"Qwen: Qwen3 VL 235B A22B Instruct","shortName":"Qwen3 VL 235B A22B Instruct","providerId":"qwen","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","contextLength":262144,"maxOutputTokens":32768,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.21,"outputPerMTok":1.9,"cachedInputPerMTok":0.09999999999999999,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-vl-235b-a22b-instruct"},"blendedPerMTok":0.6325,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"Qwen/Qwen3-VL-235B-A22B-Instruct","benchmarks":null,"releasedAt":"2025-09-23T23:04:47.000Z","isFree":false},{"id":"qwen/qwen3-max","slug":"qwen-qwen3-max","name":"Qwen: Qwen3 Max","shortName":"Qwen3 Max","providerId":"qwen","description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","contextLength":262144,"maxOutputTokens":65536,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.78,"outputPerMTok":3.9,"cachedInputPerMTok":0.156,"cacheWritePerMTok":0.975,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-max"},"blendedPerMTok":1.56,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-06-30","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-09-23T21:26:48.000Z","isFree":false},{"id":"qwen/qwen3-coder-plus","slug":"qwen-qwen3-coder-plus","name":"Qwen: Qwen3 Coder Plus","shortName":"Qwen3 Coder Plus","providerId":"qwen","description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","contextLength":1000000,"maxOutputTokens":65536,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.65,"outputPerMTok":3.25,"cachedInputPerMTok":0.13,"cacheWritePerMTok":0.8125,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-coder-plus"},"blendedPerMTok":1.3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-06-30","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-09-23T21:25:07.000Z","isFree":false},{"id":"deepseek/deepseek-v3.1-terminus","slug":"deepseek-deepseek-v3-1-terminus","name":"DeepSeek: DeepSeek V3.1 Terminus","shortName":"DeepSeek V3.1 Terminus","providerId":"deepseek","description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","contextLength":163840,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.27,"outputPerMTok":1,"cachedInputPerMTok":0.135,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v3.1-terminus"},"blendedPerMTok":0.4525,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V3.1-Terminus","benchmarks":{"intelligence_index":15.4,"coding_index":43.5,"agentic_index":8.9},"releasedAt":"2025-09-22T13:37:55.000Z","isFree":false},{"id":"qwen/qwen3-coder-flash","slug":"qwen-qwen3-coder-flash","name":"Qwen: Qwen3 Coder Flash","shortName":"Qwen3 Coder Flash","providerId":"qwen","description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","contextLength":1000000,"maxOutputTokens":65536,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.195,"outputPerMTok":0.975,"cachedInputPerMTok":0.039,"cacheWritePerMTok":0.24375,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-coder-flash"},"blendedPerMTok":0.39,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-06-30","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-09-17T13:25:36.000Z","isFree":false},{"id":"qwen/qwen3-next-80b-a3b-thinking","slug":"qwen-qwen3-next-80b-a3b-thinking","name":"Qwen: Qwen3 Next 80B A3B Thinking","shortName":"Qwen3 Next 80B A3B Thinking","providerId":"qwen","description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":1.2,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-next-80b-a3b-thinking"},"blendedPerMTok":0.4125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-09-30","openWeights":true,"huggingFaceId":"Qwen/Qwen3-Next-80B-A3B-Thinking","benchmarks":{"coding_index":17.4},"releasedAt":"2025-09-11T17:38:04.000Z","isFree":false},{"id":"qwen/qwen3-next-80b-a3b-instruct","slug":"qwen-qwen3-next-80b-a3b-instruct","name":"Qwen: Qwen3 Next 80B A3B Instruct","shortName":"Qwen3 Next 80B A3B Instruct","providerId":"qwen","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","contextLength":262144,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.09,"outputPerMTok":1.1,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-next-80b-a3b-instruct"},"blendedPerMTok":0.3425,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-09-30","openWeights":true,"huggingFaceId":"Qwen/Qwen3-Next-80B-A3B-Instruct","benchmarks":null,"releasedAt":"2025-09-11T17:36:53.000Z","isFree":false},{"id":"qwen/qwen-plus-2025-07-28","slug":"qwen-qwen-plus-2025-07-28","name":"Qwen: Qwen Plus 0728","shortName":"Qwen Plus 0728","providerId":"qwen","description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","contextLength":1000000,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.26,"outputPerMTok":0.78,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen-plus-2025-07-28"},"blendedPerMTok":0.39,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-09-08T16:06:39.000Z","isFree":false},{"id":"moonshotai/kimi-k2-0905","slug":"moonshotai-kimi-k2-0905","name":"MoonshotAI: Kimi K2 0905","shortName":"Kimi K2 0905","providerId":"moonshotai","description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","contextLength":262144,"maxOutputTokens":98304,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.6,"outputPerMTok":2.5,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2-0905"},"blendedPerMTok":1.075,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-12-31","openWeights":true,"huggingFaceId":"moonshotai/Kimi-K2-Instruct-0905","benchmarks":null,"releasedAt":"2025-09-04T21:25:47.000Z","isFree":false},{"id":"qwen/qwen3-30b-a3b-thinking-2507","slug":"qwen-qwen3-30b-a3b-thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","shortName":"Qwen3 30B A3B Thinking 2507","providerId":"qwen","description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","contextLength":81920,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":2.4,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-30b-a3b-thinking-2507"},"blendedPerMTok":0.75,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-06-30","openWeights":true,"huggingFaceId":"Qwen/Qwen3-30B-A3B-Thinking-2507","benchmarks":{"intelligence_index":9.8,"coding_index":12.1,"agentic_index":0.9},"releasedAt":"2025-08-28T16:39:52.000Z","isFree":false},{"id":"nousresearch/hermes-4-405b","slug":"nousresearch-hermes-4-405b","name":"Nous: Hermes 4 405B","shortName":"Hermes 4 405B","providerId":"nousresearch","description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":1,"outputPerMTok":3,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/nousresearch/hermes-4-405b"},"blendedPerMTok":1.5,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-08-31","openWeights":true,"huggingFaceId":"NousResearch/Hermes-4-405B","benchmarks":null,"releasedAt":"2025-08-26T19:11:03.000Z","isFree":false},{"id":"deepseek/deepseek-chat-v3.1","slug":"deepseek-deepseek-chat-v3-1","name":"DeepSeek: DeepSeek V3.1","shortName":"DeepSeek V3.1","providerId":"deepseek","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","contextLength":163840,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.25,"outputPerMTok":0.95,"cachedInputPerMTok":0.13,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-chat-v3.1"},"blendedPerMTok":0.425,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V3.1","benchmarks":null,"releasedAt":"2025-08-21T12:33:48.000Z","isFree":false},{"id":"mistralai/mistral-medium-3.1","slug":"mistralai-mistral-medium-3-1","name":"Mistral: Mistral Medium 3.1","shortName":"Mistral Medium 3.1","providerId":"mistralai","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","contextLength":131072,"maxOutputTokens":104857,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.39999999999999997,"outputPerMTok":2,"cachedInputPerMTok":0.04,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-medium-3.1"},"blendedPerMTok":0.8,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2025-06-30","openWeights":false,"huggingFaceId":"","benchmarks":{"coding_index":20.5,"agentic_index":3.1},"releasedAt":"2025-08-13T14:33:59.000Z","isFree":false},{"id":"z-ai/glm-4.5v","slug":"z-ai-glm-4-5v","name":"Z.ai: GLM 4.5V","shortName":"GLM 4.5V","providerId":"z-ai","description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","contextLength":65536,"maxOutputTokens":16384,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.6,"outputPerMTok":1.7999999999999998,"cachedInputPerMTok":0.11,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-4.5v"},"blendedPerMTok":0.8999999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-12-31","openWeights":true,"huggingFaceId":"zai-org/GLM-4.5V","benchmarks":null,"releasedAt":"2025-08-11T14:24:48.000Z","isFree":false},{"id":"openai/gpt-5","slug":"openai-gpt-5","name":"OpenAI: GPT-5","shortName":"GPT-5","providerId":"openai","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":10,"cachedInputPerMTok":0.125,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5"},"blendedPerMTok":3.4375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2024-09-30","openWeights":false,"huggingFaceId":"","benchmarks":{"coding_index":37.8},"releasedAt":"2025-08-07T17:23:33.000Z","isFree":false},{"id":"openai/gpt-5-mini","slug":"openai-gpt-5-mini","name":"OpenAI: GPT-5 Mini","shortName":"GPT-5 Mini","providerId":"openai","description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.25,"outputPerMTok":2,"cachedInputPerMTok":0.024999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5-mini"},"blendedPerMTok":0.6875,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2024-05-31","openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":17.4,"coding_index":15.6,"agentic_index":8.9},"releasedAt":"2025-08-07T17:23:27.000Z","isFree":false},{"id":"openai/gpt-5-nano","slug":"openai-gpt-5-nano","name":"OpenAI: GPT-5 Nano","shortName":"GPT-5 Nano","providerId":"openai","description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","contextLength":400000,"maxOutputTokens":128000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.049999999999999996,"outputPerMTok":0.39999999999999997,"cachedInputPerMTok":0.005,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-5-nano"},"blendedPerMTok":0.13749999999999998,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2024-05-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-08-07T17:23:22.000Z","isFree":false},{"id":"openai/gpt-oss-120b","slug":"openai-gpt-oss-120b","name":"OpenAI: gpt-oss-120b","shortName":"gpt-oss-120b","providerId":"openai","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.037,"outputPerMTok":0.16999999999999998,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-oss-120b"},"blendedPerMTok":0.07024999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":true},"knowledgeCutoff":"2024-06-30","openWeights":true,"huggingFaceId":"openai/gpt-oss-120b","benchmarks":{"intelligence_index":12.3,"coding_index":30.4,"agentic_index":6.2},"releasedAt":"2025-08-05T17:17:11.000Z","isFree":false},{"id":"openai/gpt-oss-20b","slug":"openai-gpt-oss-20b","name":"OpenAI: gpt-oss-20b","shortName":"gpt-oss-20b","providerId":"openai","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.03,"outputPerMTok":0.13,"cachedInputPerMTok":0.03,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-oss-20b"},"blendedPerMTok":0.055,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2024-06-30","openWeights":true,"huggingFaceId":"openai/gpt-oss-20b","benchmarks":{"intelligence_index":9,"coding_index":20.7,"agentic_index":1.4},"releasedAt":"2025-08-05T17:17:09.000Z","isFree":false},{"id":"anthropic/claude-opus-4.1","slug":"anthropic-claude-opus-4-1","name":"Anthropic: Claude Opus 4.1","shortName":"Claude Opus 4.1","providerId":"anthropic","description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","contextLength":200000,"maxOutputTokens":32000,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":15,"outputPerMTok":75,"cachedInputPerMTok":1.5,"cacheWritePerMTok":18.75,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.1"},"blendedPerMTok":30,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2025-01-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-08-05T16:33:11.000Z","isFree":false},{"id":"mistralai/codestral-2508","slug":"mistralai-codestral-2508","name":"Mistral: Codestral 2508","shortName":"Codestral 2508","providerId":"mistralai","description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","contextLength":256000,"maxOutputTokens":204800,"inputModalities":["text","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":0.8999999999999999,"cachedInputPerMTok":0.03,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/codestral-2508"},"blendedPerMTok":0.44999999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2025-03-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-08-01T20:20:30.000Z","isFree":false},{"id":"qwen/qwen3-coder-30b-a3b-instruct","slug":"qwen-qwen3-coder-30b-a3b-instruct","name":"Qwen: Qwen3 Coder 30B A3B Instruct","shortName":"Qwen3 Coder 30B A3B Instruct","providerId":"qwen","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.07,"outputPerMTok":0.28,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-coder-30b-a3b-instruct"},"blendedPerMTok":0.12250000000000001,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-06-30","openWeights":true,"huggingFaceId":"Qwen/Qwen3-Coder-30B-A3B-Instruct","benchmarks":null,"releasedAt":"2025-07-31T14:32:59.000Z","isFree":false},{"id":"qwen/qwen3-30b-a3b-instruct-2507","slug":"qwen-qwen3-30b-a3b-instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","shortName":"Qwen3 30B A3B Instruct 2507","providerId":"qwen","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","contextLength":262144,"maxOutputTokens":32000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.04815,"outputPerMTok":0.19305,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-30b-a3b-instruct-2507"},"blendedPerMTok":0.084375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-06-30","openWeights":true,"huggingFaceId":"Qwen/Qwen3-30B-A3B-Instruct-2507","benchmarks":null,"releasedAt":"2025-07-29T16:36:05.000Z","isFree":false},{"id":"z-ai/glm-4.5","slug":"z-ai-glm-4-5","name":"Z.ai: GLM 4.5","shortName":"GLM 4.5","providerId":"z-ai","description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","contextLength":131072,"maxOutputTokens":98304,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.6,"outputPerMTok":2.2,"cachedInputPerMTok":0.11,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-4.5"},"blendedPerMTok":1,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-12-31","openWeights":true,"huggingFaceId":"zai-org/GLM-4.5","benchmarks":null,"releasedAt":"2025-07-25T19:22:27.000Z","isFree":false},{"id":"z-ai/glm-4.5-air","slug":"z-ai-glm-4-5-air","name":"Z.ai: GLM 4.5 Air","shortName":"GLM 4.5 Air","providerId":"z-ai","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","contextLength":131072,"maxOutputTokens":98304,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.13,"outputPerMTok":0.85,"cachedInputPerMTok":0.024999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/z-ai/glm-4.5-air"},"blendedPerMTok":0.31,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-12-31","openWeights":true,"huggingFaceId":"zai-org/GLM-4.5-Air","benchmarks":null,"releasedAt":"2025-07-25T19:20:58.000Z","isFree":false},{"id":"qwen/qwen3-235b-a22b-thinking-2507","slug":"qwen-qwen3-235b-a22b-thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","shortName":"Qwen3 235B A22B Thinking 2507","providerId":"qwen","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.22999999999999998,"outputPerMTok":2.3,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-235b-a22b-thinking-2507"},"blendedPerMTok":0.7474999999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-06-30","openWeights":true,"huggingFaceId":"Qwen/Qwen3-235B-A22B-Thinking-2507","benchmarks":{"intelligence_index":12.7,"coding_index":22.1,"agentic_index":1.3},"releasedAt":"2025-07-25T13:19:17.000Z","isFree":false},{"id":"qwen/qwen3-coder","slug":"qwen-qwen3-coder","name":"Qwen: Qwen3 Coder 480B A35B","shortName":"Qwen3 Coder 480B A35B","providerId":"qwen","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","contextLength":262144,"maxOutputTokens":65536,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":1,"cachedInputPerMTok":0.09999999999999999,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-coder"},"blendedPerMTok":0.475,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-06-30","openWeights":true,"huggingFaceId":"Qwen/Qwen3-Coder-480B-A35B-Instruct","benchmarks":null,"releasedAt":"2025-07-23T00:29:06.000Z","isFree":false},{"id":"bytedance/ui-tars-1.5-7b","slug":"bytedance-ui-tars-1-5-7b","name":"ByteDance: UI-TARS 7B ","shortName":"UI-TARS 7B ","providerId":"bytedance","description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","contextLength":128000,"maxOutputTokens":2048,"inputModalities":["image","text"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.19999999999999998,"cachedInputPerMTok":0.09999999999999999,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/bytedance/ui-tars-1.5-7b"},"blendedPerMTok":0.125,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-01-31","openWeights":true,"huggingFaceId":"ByteDance-Seed/UI-TARS-1.5-7B","benchmarks":null,"releasedAt":"2025-07-22T17:24:16.000Z","isFree":false},{"id":"google/gemini-2.5-flash-lite","slug":"google-gemini-2-5-flash-lite","name":"Google: Gemini 2.5 Flash Lite","shortName":"Gemini 2.5 Flash Lite","providerId":"google","description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","contextLength":1048576,"maxOutputTokens":65535,"inputModalities":["text","image","file","audio","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.39999999999999997,"cachedInputPerMTok":0.01,"cacheWritePerMTok":0.0833333333333333,"perRequest":null,"perImage":1e-7,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-2.5-flash-lite"},"blendedPerMTok":0.175,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2025-01-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-07-22T16:04:36.000Z","isFree":false},{"id":"qwen/qwen3-235b-a22b-2507","slug":"qwen-qwen3-235b-a22b-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","shortName":"Qwen3 235B A22B Instruct 2507","providerId":"qwen","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","contextLength":262144,"maxOutputTokens":235929,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.0875,"outputPerMTok":0.35,"cachedInputPerMTok":0.0175,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-235b-a22b-2507"},"blendedPerMTok":0.15312499999999998,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-06-30","openWeights":true,"huggingFaceId":"Qwen/Qwen3-235B-A22B-Instruct-2507","benchmarks":null,"releasedAt":"2025-07-21T17:39:15.000Z","isFree":false},{"id":"moonshotai/kimi-k2","slug":"moonshotai-kimi-k2","name":"MoonshotAI: Kimi K2 0711","shortName":"Kimi K2 0711","providerId":"moonshotai","description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","contextLength":131072,"maxOutputTokens":98304,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.5700000000000001,"outputPerMTok":2.3,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2"},"blendedPerMTok":1.0025,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-12-31","openWeights":true,"huggingFaceId":"moonshotai/Kimi-K2-Instruct","benchmarks":null,"releasedAt":"2025-07-11T19:47:32.000Z","isFree":false},{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition","slug":"cognitivecomputations-dolphin-mistral-24b-venice-edition","name":"Venice: Uncensored","shortName":"Uncensored","providerId":"cognitivecomputations","description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","contextLength":128000,"maxOutputTokens":8192,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":0.8999999999999999,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/cognitivecomputations/dolphin-mistral-24b-venice-edition"},"blendedPerMTok":0.375,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-04-30","openWeights":true,"huggingFaceId":"cognitivecomputations/Dolphin-Mistral-24B-Venice-Edition","benchmarks":null,"releasedAt":"2025-07-09T21:02:46.000Z","isFree":false},{"id":"tencent/hunyuan-a13b-instruct","slug":"tencent-hunyuan-a13b-instruct","name":"Tencent: Hunyuan A13B Instruct","shortName":"Hunyuan A13B Instruct","providerId":"tencent","description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.14,"outputPerMTok":0.5700000000000001,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/tencent/hunyuan-a13b-instruct"},"blendedPerMTok":0.24750000000000003,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"tencent/Hunyuan-A13B-Instruct","benchmarks":null,"releasedAt":"2025-07-08T15:14:24.000Z","isFree":false},{"id":"morph/morph-v3-large","slug":"morph-morph-v3-large","name":"Morph: Morph V3 Large","shortName":"Morph V3 Large","providerId":"morph","description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","contextLength":262144,"maxOutputTokens":131072,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.8999999999999999,"outputPerMTok":1.9,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/morph/morph-v3-large"},"blendedPerMTok":1.15,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-07-07T17:54:18.000Z","isFree":false},{"id":"morph/morph-v3-fast","slug":"morph-morph-v3-fast","name":"Morph: Morph V3 Fast","shortName":"Morph V3 Fast","providerId":"morph","description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","contextLength":81920,"maxOutputTokens":38000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.7999999999999999,"outputPerMTok":1.2,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/morph/morph-v3-fast"},"blendedPerMTok":0.8999999999999999,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-07-07T17:40:02.000Z","isFree":false},{"id":"baidu/ernie-4.5-vl-424b-a47b","slug":"baidu-ernie-4-5-vl-424b-a47b","name":"Baidu: ERNIE 4.5 VL 424B A47B ","shortName":"ERNIE 4.5 VL 424B A47B ","providerId":"baidu","description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","contextLength":123000,"maxOutputTokens":16000,"inputModalities":["image","text"],"outputModalities":["text"],"price":{"inputPerMTok":0.42,"outputPerMTok":1.25,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/baidu/ernie-4.5-vl-424b-a47b"},"blendedPerMTok":0.6275,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"baidu/ERNIE-4.5-VL-424B-A47B-PT","benchmarks":null,"releasedAt":"2025-06-30T16:28:23.000Z","isFree":false},{"id":"mistralai/mistral-small-3.2-24b-instruct","slug":"mistralai-mistral-small-3-2-24b-instruct","name":"Mistral: Mistral Small 3.2 24B","shortName":"Mistral Small 3.2 24B","providerId":"mistralai","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","contextLength":256000,"maxOutputTokens":16384,"inputModalities":["image","text"],"outputModalities":["text"],"price":{"inputPerMTok":0.075,"outputPerMTok":0.19999999999999998,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-small-3.2-24b-instruct"},"blendedPerMTok":0.10624999999999998,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-10-31","openWeights":true,"huggingFaceId":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","benchmarks":null,"releasedAt":"2025-06-20T18:10:16.000Z","isFree":false},{"id":"minimax/minimax-m1","slug":"minimax-minimax-m1","name":"MiniMax: MiniMax M1","shortName":"MiniMax M1","providerId":"minimax","description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","contextLength":1000000,"maxOutputTokens":40000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.55,"outputPerMTok":2.2,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/minimax/minimax-m1"},"blendedPerMTok":0.9625000000000001,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-06-30","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-06-17T22:46:54.000Z","isFree":false},{"id":"google/gemini-2.5-flash","slug":"google-gemini-2-5-flash","name":"Google: Gemini 2.5 Flash","shortName":"Gemini 2.5 Flash","providerId":"google","description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","contextLength":1048576,"maxOutputTokens":65535,"inputModalities":["file","image","text","audio","video"],"outputModalities":["text"],"price":{"inputPerMTok":0.3,"outputPerMTok":2.5,"cachedInputPerMTok":0.03,"cacheWritePerMTok":0.0833333333333333,"perRequest":null,"perImage":3e-7,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-2.5-flash"},"blendedPerMTok":0.85,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2025-01-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-06-17T15:01:28.000Z","isFree":false},{"id":"google/gemini-2.5-pro","slug":"google-gemini-2-5-pro","name":"Google: Gemini 2.5 Pro","shortName":"Gemini 2.5 Pro","providerId":"google","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["text","image","file","audio","video"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":10,"cachedInputPerMTok":0.125,"cacheWritePerMTok":0.375,"perRequest":null,"perImage":0.00000125,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-2.5-pro"},"blendedPerMTok":3.4375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2025-01-31","openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":16.7,"coding_index":33.3,"agentic_index":3.5},"releasedAt":"2025-06-17T14:12:24.000Z","isFree":false},{"id":"openai/o3-pro","slug":"openai-o3-pro","name":"OpenAI: o3 Pro","shortName":"o3 Pro","providerId":"openai","description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","contextLength":200000,"maxOutputTokens":100000,"inputModalities":["text","file","image"],"outputModalities":["text"],"price":{"inputPerMTok":20,"outputPerMTok":80,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/o3-pro"},"blendedPerMTok":35,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-06-30","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-06-10T23:32:32.000Z","isFree":false},{"id":"google/gemini-2.5-pro-preview","slug":"google-gemini-2-5-pro-preview","name":"Google: Gemini 2.5 Pro Preview 06-05","shortName":"Gemini 2.5 Pro Preview 06-05","providerId":"google","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","contextLength":1048576,"maxOutputTokens":65536,"inputModalities":["file","image","text","audio"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":10,"cachedInputPerMTok":0.125,"cacheWritePerMTok":0.375,"perRequest":null,"perImage":0.00000125,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-2.5-pro-preview"},"blendedPerMTok":3.4375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-01-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-06-05T15:27:37.000Z","isFree":false},{"id":"deepseek/deepseek-r1-0528","slug":"deepseek-deepseek-r1-0528","name":"DeepSeek: R1 0528","shortName":"R1 0528","providerId":"deepseek","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","contextLength":163840,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.5,"outputPerMTok":2.1500000000000004,"cachedInputPerMTok":0.35,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-r1-0528"},"blendedPerMTok":0.9125000000000001,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-R1-0528","benchmarks":null,"releasedAt":"2025-05-28T17:59:30.000Z","isFree":false},{"id":"anthropic/claude-opus-4","slug":"anthropic-claude-opus-4","name":"Anthropic: Claude Opus 4","shortName":"Claude Opus 4","providerId":"anthropic","description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","contextLength":200000,"maxOutputTokens":32000,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":15,"outputPerMTok":75,"cachedInputPerMTok":1.5,"cacheWritePerMTok":18.75,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4"},"blendedPerMTok":30,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-01-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-05-22T16:27:25.000Z","isFree":false},{"id":"anthropic/claude-sonnet-4","slug":"anthropic-claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","shortName":"Claude Sonnet 4","providerId":"anthropic","description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","contextLength":1000000,"maxOutputTokens":64000,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":3,"outputPerMTok":15,"cachedInputPerMTok":0.3,"cacheWritePerMTok":3.75,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-sonnet-4"},"blendedPerMTok":6,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-01-31","openWeights":false,"huggingFaceId":"","benchmarks":{"coding_index":37.6},"releasedAt":"2025-05-22T16:12:51.000Z","isFree":false},{"id":"mistralai/mistral-medium-3","slug":"mistralai-mistral-medium-3","name":"Mistral: Mistral Medium 3","shortName":"Mistral Medium 3","providerId":"mistralai","description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","contextLength":131072,"maxOutputTokens":104857,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.39999999999999997,"outputPerMTok":2,"cachedInputPerMTok":0.04,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-medium-3"},"blendedPerMTok":0.8,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-05-07T14:15:41.000Z","isFree":false},{"id":"google/gemini-2.5-pro-preview-05-06","slug":"google-gemini-2-5-pro-preview-05-06","name":"Google: Gemini 2.5 Pro Preview 05-06","shortName":"Gemini 2.5 Pro Preview 05-06","providerId":"google","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","contextLength":1048576,"maxOutputTokens":65535,"inputModalities":["text","image","file","audio","video"],"outputModalities":["text"],"price":{"inputPerMTok":1.25,"outputPerMTok":10,"cachedInputPerMTok":0.125,"cacheWritePerMTok":0.375,"perRequest":null,"perImage":0.00000125,"searchPerRequest":0.014,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemini-2.5-pro-preview-05-06"},"blendedPerMTok":3.4375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-01-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-05-07T00:41:53.000Z","isFree":false},{"id":"meta-llama/llama-guard-4-12b","slug":"meta-llama-llama-guard-4-12b","name":"Meta: Llama Guard 4 12B","shortName":"Llama Guard 4 12B","providerId":"meta-llama","description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","contextLength":163840,"maxOutputTokens":16384,"inputModalities":["image","text"],"outputModalities":["text"],"price":{"inputPerMTok":0.18,"outputPerMTok":0.18,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta-llama/llama-guard-4-12b"},"blendedPerMTok":0.18,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-08-31","openWeights":true,"huggingFaceId":"meta-llama/Llama-Guard-4-12B","benchmarks":null,"releasedAt":"2025-04-30T01:06:33.000Z","isFree":false},{"id":"qwen/qwen3-30b-a3b","slug":"qwen-qwen3-30b-a3b","name":"Qwen: Qwen3 30B A3B","shortName":"Qwen3 30B A3B","providerId":"qwen","description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.12,"outputPerMTok":0.5,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-30b-a3b"},"blendedPerMTok":0.215,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"Qwen/Qwen3-30B-A3B","benchmarks":null,"releasedAt":"2025-04-28T22:16:44.000Z","isFree":false},{"id":"qwen/qwen3-8b","slug":"qwen-qwen3-8b","name":"Qwen: Qwen3 8B","shortName":"Qwen3 8B","providerId":"qwen","description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","contextLength":131072,"maxOutputTokens":8192,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.117,"outputPerMTok":0.45499999999999996,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-8b"},"blendedPerMTok":0.2015,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"Qwen/Qwen3-8B","benchmarks":{"intelligence_index":5.2,"coding_index":9,"agentic_index":0.8},"releasedAt":"2025-04-28T21:43:52.000Z","isFree":false},{"id":"qwen/qwen3-14b","slug":"qwen-qwen3-14b","name":"Qwen: Qwen3 14B","shortName":"Qwen3 14B","providerId":"qwen","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.12,"outputPerMTok":0.24,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-14b"},"blendedPerMTok":0.15,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"Qwen/Qwen3-14B","benchmarks":{"intelligence_index":6.4,"coding_index":13.8,"agentic_index":0.9},"releasedAt":"2025-04-28T21:41:18.000Z","isFree":false},{"id":"qwen/qwen3-32b","slug":"qwen-qwen3-32b","name":"Qwen: Qwen3 32B","shortName":"Qwen3 32B","providerId":"qwen","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.08,"outputPerMTok":0.28,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-32b"},"blendedPerMTok":0.13,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"Qwen/Qwen3-32B","benchmarks":{"intelligence_index":7.2,"coding_index":15.3,"agentic_index":0.9},"releasedAt":"2025-04-28T21:32:25.000Z","isFree":false},{"id":"qwen/qwen3-235b-a22b","slug":"qwen-qwen3-235b-a22b","name":"Qwen: Qwen3 235B A22B","shortName":"Qwen3 235B A22B","providerId":"qwen","description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","contextLength":131072,"maxOutputTokens":8192,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.45499999999999996,"outputPerMTok":1.8199999999999998,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen3-235b-a22b"},"blendedPerMTok":0.7962499999999999,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":true,"huggingFaceId":"Qwen/Qwen3-235B-A22B","benchmarks":null,"releasedAt":"2025-04-28T21:29:17.000Z","isFree":false},{"id":"openai/o4-mini-high","slug":"openai-o4-mini-high","name":"OpenAI: o4 Mini High","shortName":"o4 Mini High","providerId":"openai","description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","contextLength":200000,"maxOutputTokens":100000,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":1.1,"outputPerMTok":4.4,"cachedInputPerMTok":0.275,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/o4-mini-high"},"blendedPerMTok":1.9250000000000003,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-06-30","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-04-16T17:23:32.000Z","isFree":false},{"id":"openai/o3","slug":"openai-o3","name":"OpenAI: o3","shortName":"o3","providerId":"openai","description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","contextLength":200000,"maxOutputTokens":100000,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":8,"cachedInputPerMTok":0.5,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/o3"},"blendedPerMTok":3.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2024-06-30","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-04-16T17:10:57.000Z","isFree":false},{"id":"openai/o4-mini","slug":"openai-o4-mini","name":"OpenAI: o4 Mini","shortName":"o4 Mini","providerId":"openai","description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","contextLength":200000,"maxOutputTokens":100000,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":1.1,"outputPerMTok":4.4,"cachedInputPerMTok":0.275,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/o4-mini"},"blendedPerMTok":1.9250000000000003,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2024-06-30","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-04-16T16:29:02.000Z","isFree":false},{"id":"openai/gpt-4.1","slug":"openai-gpt-4-1","name":"OpenAI: GPT-4.1","shortName":"GPT-4.1","providerId":"openai","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","contextLength":1047576,"maxOutputTokens":32768,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":8,"cachedInputPerMTok":0.5,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4.1"},"blendedPerMTok":3.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2024-06-30","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-04-14T17:23:05.000Z","isFree":false},{"id":"openai/gpt-4.1-mini","slug":"openai-gpt-4-1-mini","name":"OpenAI: GPT-4.1 Mini","shortName":"GPT-4.1 Mini","providerId":"openai","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","contextLength":1047576,"maxOutputTokens":32768,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.39999999999999997,"outputPerMTok":1.5999999999999999,"cachedInputPerMTok":0.09999999999999999,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4.1-mini"},"blendedPerMTok":0.7,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2024-06-30","openWeights":false,"huggingFaceId":"","benchmarks":{"coding_index":20.2},"releasedAt":"2025-04-14T17:23:01.000Z","isFree":false},{"id":"openai/gpt-4.1-nano","slug":"openai-gpt-4-1-nano","name":"OpenAI: GPT-4.1 Nano","shortName":"GPT-4.1 Nano","providerId":"openai","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","contextLength":1047576,"maxOutputTokens":32768,"inputModalities":["image","text","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.39999999999999997,"cachedInputPerMTok":0.024999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4.1-nano"},"blendedPerMTok":0.175,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2024-06-30","openWeights":false,"huggingFaceId":"","benchmarks":{"coding_index":11.1},"releasedAt":"2025-04-14T17:22:49.000Z","isFree":false},{"id":"meta-llama/llama-4-maverick","slug":"meta-llama-llama-4-maverick","name":"Meta: Llama 4 Maverick","shortName":"Llama 4 Maverick","providerId":"meta-llama","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","contextLength":1048576,"maxOutputTokens":115200,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":0.696,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta-llama/llama-4-maverick"},"blendedPerMTok":0.32399999999999995,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-08-31","openWeights":true,"huggingFaceId":"meta-llama/Llama-4-Maverick-17B-128E-Instruct","benchmarks":{"intelligence_index":9.3,"coding_index":16.3,"agentic_index":0.6},"releasedAt":"2025-04-05T19:37:02.000Z","isFree":false},{"id":"meta-llama/llama-4-scout","slug":"meta-llama-llama-4-scout","name":"Meta: Llama 4 Scout","shortName":"Llama 4 Scout","providerId":"meta-llama","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","contextLength":1310720,"maxOutputTokens":16384,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.3,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta-llama/llama-4-scout"},"blendedPerMTok":0.15,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-08-31","openWeights":true,"huggingFaceId":"meta-llama/Llama-4-Scout-17B-16E-Instruct","benchmarks":{"intelligence_index":6.5,"coding_index":8.2,"agentic_index":0.5},"releasedAt":"2025-04-05T19:31:59.000Z","isFree":false},{"id":"deepseek/deepseek-chat-v3-0324","slug":"deepseek-deepseek-chat-v3-0324","name":"DeepSeek: DeepSeek V3 0324","shortName":"DeepSeek V3 0324","providerId":"deepseek","description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","contextLength":163840,"maxOutputTokens":147456,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.25,"outputPerMTok":1,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-chat-v3-0324"},"blendedPerMTok":0.4375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-07-31","openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V3-0324","benchmarks":{"intelligence_index":9.7,"coding_index":21.2,"agentic_index":0.8},"releasedAt":"2025-03-24T13:59:15.000Z","isFree":false},{"id":"openai/o1-pro","slug":"openai-o1-pro","name":"OpenAI: o1-pro","shortName":"o1-pro","providerId":"openai","description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","contextLength":200000,"maxOutputTokens":100000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":150,"outputPerMTok":600,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/o1-pro"},"blendedPerMTok":262.5,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-10-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-03-19T22:26:51.000Z","isFree":false},{"id":"mistralai/mistral-small-3.1-24b-instruct","slug":"mistralai-mistral-small-3-1-24b-instruct","name":"Mistral: Mistral Small 3.1 24B","shortName":"Mistral Small 3.1 24B","providerId":"mistralai","description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","contextLength":128000,"maxOutputTokens":102400,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.351,"outputPerMTok":0.5549999999999999,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-small-3.1-24b-instruct"},"blendedPerMTok":0.40199999999999997,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-10-31","openWeights":true,"huggingFaceId":"mistralai/Mistral-Small-3.1-24B-Instruct-2503","benchmarks":null,"releasedAt":"2025-03-17T19:15:37.000Z","isFree":false},{"id":"google/gemma-3-4b-it","slug":"google-gemma-3-4b-it","name":"Google: Gemma 3 4B","shortName":"Gemma 3 4B","providerId":"google","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.049999999999999996,"outputPerMTok":0.09999999999999999,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemma-3-4b-it"},"blendedPerMTok":0.0625,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-08-31","openWeights":true,"huggingFaceId":"google/gemma-3-4b-it","benchmarks":{"coding_index":2.7},"releasedAt":"2025-03-13T22:38:30.000Z","isFree":false},{"id":"google/gemma-3-12b-it","slug":"google-gemma-3-12b-it","name":"Google: Gemma 3 12B","shortName":"Gemma 3 12B","providerId":"google","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.049999999999999996,"outputPerMTok":0.15,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemma-3-12b-it"},"blendedPerMTok":0.075,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-08-31","openWeights":true,"huggingFaceId":"google/gemma-3-12b-it","benchmarks":{"intelligence_index":3.8,"coding_index":5.8,"agentic_index":0.1},"releasedAt":"2025-03-13T21:50:25.000Z","isFree":false},{"id":"cohere/command-a","slug":"cohere-command-a","name":"Cohere: Command A","shortName":"Command A","providerId":"cohere","description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","contextLength":256000,"maxOutputTokens":8192,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":2.5,"outputPerMTok":10,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/cohere/command-a"},"blendedPerMTok":4.375,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-08-31","openWeights":true,"huggingFaceId":"CohereForAI/c4ai-command-a-03-2025","benchmarks":{"intelligence_index":13.9,"coding_index":27.8,"agentic_index":3.6},"releasedAt":"2025-03-13T19:32:22.000Z","isFree":false},{"id":"rekaai/reka-flash-3","slug":"rekaai-reka-flash-3","name":"Reka Flash 3","shortName":"Reka Flash 3","providerId":"rekaai","description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","contextLength":65536,"maxOutputTokens":58982,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.19999999999999998,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/rekaai/reka-flash-3"},"blendedPerMTok":0.125,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2025-01-31","openWeights":true,"huggingFaceId":"RekaAI/reka-flash-3","benchmarks":null,"releasedAt":"2025-03-12T20:53:33.000Z","isFree":false},{"id":"google/gemma-3-27b-it","slug":"google-gemma-3-27b-it","name":"Google: Gemma 3 27B","shortName":"Gemma 3 27B","providerId":"google","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.08,"outputPerMTok":0.44999999999999996,"cachedInputPerMTok":0.04,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemma-3-27b-it"},"blendedPerMTok":0.1725,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-08-31","openWeights":true,"huggingFaceId":"google/gemma-3-27b-it","benchmarks":{"intelligence_index":4.9,"coding_index":10.1,"agentic_index":0.1},"releasedAt":"2025-03-12T05:12:39.000Z","isFree":false},{"id":"thedrummer/skyfall-36b-v2","slug":"thedrummer-skyfall-36b-v2","name":"TheDrummer: Skyfall 36B V2","shortName":"Skyfall 36B V2","providerId":"thedrummer","description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","contextLength":32768,"maxOutputTokens":29491,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.55,"outputPerMTok":0.7999999999999999,"cachedInputPerMTok":0.25,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/thedrummer/skyfall-36b-v2"},"blendedPerMTok":0.6125,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-06-30","openWeights":true,"huggingFaceId":"TheDrummer/Skyfall-36B-v2","benchmarks":null,"releasedAt":"2025-03-10T19:56:06.000Z","isFree":false},{"id":"perplexity/sonar-reasoning-pro","slug":"perplexity-sonar-reasoning-pro","name":"Perplexity: Sonar Reasoning Pro","shortName":"Sonar Reasoning Pro","providerId":"perplexity","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","contextLength":128000,"maxOutputTokens":115200,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":8,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.005,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/perplexity/sonar-reasoning-pro"},"blendedPerMTok":3.5,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-03-07T02:08:28.000Z","isFree":false},{"id":"perplexity/sonar-pro","slug":"perplexity-sonar-pro","name":"Perplexity: Sonar Pro","shortName":"Sonar Pro","providerId":"perplexity","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","contextLength":200000,"maxOutputTokens":8000,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":3,"outputPerMTok":15,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.005,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/perplexity/sonar-pro"},"blendedPerMTok":6,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-03-07T01:53:43.000Z","isFree":false},{"id":"perplexity/sonar-deep-research","slug":"perplexity-sonar-deep-research","name":"Perplexity: Sonar Deep Research","shortName":"Sonar Deep Research","providerId":"perplexity","description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","contextLength":128000,"maxOutputTokens":115200,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":8,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.005,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/perplexity/sonar-deep-research"},"blendedPerMTok":3.5,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-03-07T01:34:06.000Z","isFree":false},{"id":"mistralai/mistral-saba","slug":"mistralai-mistral-saba","name":"Mistral: Saba","shortName":"Saba","providerId":"mistralai","description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","contextLength":32768,"maxOutputTokens":26214,"inputModalities":["text","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":0.6,"cachedInputPerMTok":0.02,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-saba"},"blendedPerMTok":0.3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-09-30","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-02-17T14:40:39.000Z","isFree":false},{"id":"openai/o3-mini-high","slug":"openai-o3-mini-high","name":"OpenAI: o3 Mini High","shortName":"o3 Mini High","providerId":"openai","description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","contextLength":200000,"maxOutputTokens":100000,"inputModalities":["text","file"],"outputModalities":["text"],"price":{"inputPerMTok":1.1,"outputPerMTok":4.4,"cachedInputPerMTok":0.55,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/o3-mini-high"},"blendedPerMTok":1.9250000000000003,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2023-10-31","openWeights":false,"huggingFaceId":"","benchmarks":{"intelligence_index":11,"coding_index":16.3,"agentic_index":0.9},"releasedAt":"2025-02-12T15:03:31.000Z","isFree":false},{"id":"aion-labs/aion-rp-llama-3.1-8b","slug":"aion-labs-aion-rp-llama-3-1-8b","name":"AionLabs: Aion-RP 1.0 (8B)","shortName":"Aion-RP 1.0 (8B)","providerId":"aion-labs","description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","contextLength":32768,"maxOutputTokens":29491,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.7999999999999999,"outputPerMTok":1.5999999999999999,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/aion-labs/aion-rp-llama-3.1-8b"},"blendedPerMTok":1,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-02-04T19:18:38.000Z","isFree":false},{"id":"qwen/qwen2.5-vl-72b-instruct","slug":"qwen-qwen2-5-vl-72b-instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","shortName":"Qwen2.5 VL 72B Instruct","providerId":"qwen","description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","contextLength":128000,"maxOutputTokens":115200,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.7999999999999999,"outputPerMTok":1,"cachedInputPerMTok":0.39999999999999997,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen2.5-vl-72b-instruct"},"blendedPerMTok":0.85,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-06-30","openWeights":true,"huggingFaceId":"Qwen/Qwen2.5-VL-72B-Instruct","benchmarks":null,"releasedAt":"2025-02-01T11:45:11.000Z","isFree":false},{"id":"qwen/qwen-plus","slug":"qwen-qwen-plus","name":"Qwen: Qwen-Plus","shortName":"Qwen-Plus","providerId":"qwen","description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","contextLength":1000000,"maxOutputTokens":32768,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.26,"outputPerMTok":0.78,"cachedInputPerMTok":0.052000000000000005,"cacheWritePerMTok":0.325,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen-plus"},"blendedPerMTok":0.39,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2025-03-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-02-01T11:37:20.000Z","isFree":false},{"id":"openai/o3-mini","slug":"openai-o3-mini","name":"OpenAI: o3 Mini","shortName":"o3 Mini","providerId":"openai","description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","contextLength":200000,"maxOutputTokens":100000,"inputModalities":["text","file"],"outputModalities":["text"],"price":{"inputPerMTok":1.1,"outputPerMTok":4.4,"cachedInputPerMTok":0.55,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/o3-mini"},"blendedPerMTok":1.9250000000000003,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2023-10-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-01-31T19:28:41.000Z","isFree":false},{"id":"mistralai/mistral-small-24b-instruct-2501","slug":"mistralai-mistral-small-24b-instruct-2501","name":"Mistral: Mistral Small 3","shortName":"Mistral Small 3","providerId":"mistralai","description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","contextLength":32768,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.049999999999999996,"outputPerMTok":0.08,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-small-24b-instruct-2501"},"blendedPerMTok":0.057499999999999996,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-10-31","openWeights":true,"huggingFaceId":"mistralai/Mistral-Small-24B-Instruct-2501","benchmarks":null,"releasedAt":"2025-01-30T16:43:29.000Z","isFree":false},{"id":"perplexity/sonar","slug":"perplexity-sonar","name":"Perplexity: Sonar","shortName":"Sonar","providerId":"perplexity","description":"Sonar is lightweight, affordable, fast, and simple to use - now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","contextLength":127072,"maxOutputTokens":114364,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":1,"outputPerMTok":1,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.005,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/perplexity/sonar"},"blendedPerMTok":1,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2025-01-27T21:36:48.000Z","isFree":false},{"id":"deepseek/deepseek-r1-distill-llama-70b","slug":"deepseek-deepseek-r1-distill-llama-70b","name":"DeepSeek: R1 Distill Llama 70B","shortName":"R1 Distill Llama 70B","providerId":"deepseek","description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","contextLength":8192,"maxOutputTokens":7372,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.7999999999999999,"outputPerMTok":0.7999999999999999,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-r1-distill-llama-70b"},"blendedPerMTok":0.7999999999999999,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-07-31","openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-R1-Distill-Llama-70B","benchmarks":null,"releasedAt":"2025-01-23T20:12:49.000Z","isFree":false},{"id":"deepseek/deepseek-r1","slug":"deepseek-deepseek-r1","name":"DeepSeek: R1","shortName":"R1","providerId":"deepseek","description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","contextLength":64000,"maxOutputTokens":16000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.7,"outputPerMTok":2.5,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-r1"},"blendedPerMTok":1.15,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-07-31","openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-R1","benchmarks":{"intelligence_index":11.4,"coding_index":24.6,"agentic_index":1.1},"releasedAt":"2025-01-20T13:51:35.000Z","isFree":false},{"id":"minimax/minimax-01","slug":"minimax-minimax-01","name":"MiniMax: MiniMax-01","shortName":"MiniMax-01","providerId":"minimax","description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","contextLength":1000192,"maxOutputTokens":900172,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.19999999999999998,"outputPerMTok":1.1,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/minimax/minimax-01"},"blendedPerMTok":0.42500000000000004,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-03-31","openWeights":true,"huggingFaceId":"MiniMaxAI/MiniMax-Text-01","benchmarks":null,"releasedAt":"2025-01-15T04:31:02.000Z","isFree":false},{"id":"microsoft/phi-4","slug":"microsoft-phi-4","name":"Microsoft: Phi 4","shortName":"Phi 4","providerId":"microsoft","description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","contextLength":16384,"maxOutputTokens":14745,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.07,"outputPerMTok":0.14,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/microsoft/phi-4"},"blendedPerMTok":0.08750000000000001,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-06-30","openWeights":true,"huggingFaceId":"microsoft/phi-4","benchmarks":null,"releasedAt":"2025-01-10T06:17:52.000Z","isFree":false},{"id":"deepseek/deepseek-chat","slug":"deepseek-deepseek-chat","name":"DeepSeek: DeepSeek V3","shortName":"DeepSeek V3","providerId":"deepseek","description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","contextLength":163840,"maxOutputTokens":16000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.2574,"outputPerMTok":1.0287,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/deepseek/deepseek-chat"},"blendedPerMTok":0.450225,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-07-31","openWeights":true,"huggingFaceId":"deepseek-ai/DeepSeek-V3","benchmarks":null,"releasedAt":"2024-12-26T19:28:40.000Z","isFree":false},{"id":"sao10k/l3.3-euryale-70b","slug":"sao10k-l3-3-euryale-70b","name":"Sao10K: Llama 3.3 Euryale 70B","shortName":"Llama 3.3 Euryale 70B","providerId":"sao10k","description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.65,"outputPerMTok":0.75,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/sao10k/l3.3-euryale-70b"},"blendedPerMTok":0.675,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":true,"huggingFaceId":"Sao10K/L3.3-70B-Euryale-v2.3","benchmarks":null,"releasedAt":"2024-12-18T15:32:08.000Z","isFree":false},{"id":"openai/o1","slug":"openai-o1","name":"OpenAI: o1","shortName":"o1","providerId":"openai","description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","contextLength":200000,"maxOutputTokens":100000,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":15,"outputPerMTok":60,"cachedInputPerMTok":7.5,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/o1"},"blendedPerMTok":26.25,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2023-10-31","openWeights":false,"huggingFaceId":"","benchmarks":{"coding_index":39.7},"releasedAt":"2024-12-17T18:26:39.000Z","isFree":false},{"id":"cohere/command-r7b-12-2024","slug":"cohere-command-r7b-12-2024","name":"Cohere: Command R7B (12-2024)","shortName":"Command R7B (12-2024)","providerId":"cohere","description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","contextLength":128000,"maxOutputTokens":4000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.0375,"outputPerMTok":0.15,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/cohere/command-r7b-12-2024"},"blendedPerMTok":0.06562499999999999,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-08-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2024-12-14T06:35:52.000Z","isFree":false},{"id":"meta-llama/llama-3.3-70b-instruct","slug":"meta-llama-llama-3-3-70b-instruct","name":"Meta: Llama 3.3 70B Instruct","shortName":"Llama 3.3 70B Instruct","providerId":"meta-llama","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.32,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta-llama/llama-3.3-70b-instruct"},"blendedPerMTok":0.155,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":true,"huggingFaceId":"meta-llama/Llama-3.3-70B-Instruct","benchmarks":{"coding_index":11.9},"releasedAt":"2024-12-06T17:28:57.000Z","isFree":false},{"id":"amazon/nova-lite-v1","slug":"amazon-nova-lite-v1","name":"Amazon: Nova Lite 1.0","shortName":"Nova Lite 1.0","providerId":"amazon","description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","contextLength":300000,"maxOutputTokens":5120,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.06,"outputPerMTok":0.24,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/amazon/nova-lite-v1"},"blendedPerMTok":0.105,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-10-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2024-12-05T22:22:43.000Z","isFree":false},{"id":"amazon/nova-micro-v1","slug":"amazon-nova-micro-v1","name":"Amazon: Nova Micro 1.0","shortName":"Nova Micro 1.0","providerId":"amazon","description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","contextLength":128000,"maxOutputTokens":5120,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.035,"outputPerMTok":0.14,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/amazon/nova-micro-v1"},"blendedPerMTok":0.061250000000000006,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-10-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2024-12-05T22:20:37.000Z","isFree":false},{"id":"amazon/nova-pro-v1","slug":"amazon-nova-pro-v1","name":"Amazon: Nova Pro 1.0","shortName":"Nova Pro 1.0","providerId":"amazon","description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","contextLength":300000,"maxOutputTokens":5120,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.7999999999999999,"outputPerMTok":3.1999999999999997,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/amazon/nova-pro-v1"},"blendedPerMTok":1.4,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-10-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2024-12-05T22:05:03.000Z","isFree":false},{"id":"openai/gpt-4o-2024-11-20","slug":"openai-gpt-4o-2024-11-20","name":"OpenAI: GPT-4o (2024-11-20)","shortName":"GPT-4o (2024-11-20)","providerId":"openai","description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","contextLength":128000,"maxOutputTokens":16384,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":2.5,"outputPerMTok":10,"cachedInputPerMTok":1.25,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4o-2024-11-20"},"blendedPerMTok":4.375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2023-10-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2024-11-20T18:33:14.000Z","isFree":false},{"id":"mistralai/mistral-large-2407","slug":"mistralai-mistral-large-2407","name":"Mistral Large 2407","shortName":"Mistral Large 2407","providerId":"mistralai","description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","contextLength":131072,"maxOutputTokens":104857,"inputModalities":["text","file"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":6,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-large-2407"},"blendedPerMTok":3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-03-31","openWeights":false,"huggingFaceId":"","benchmarks":null,"releasedAt":"2024-11-19T01:06:55.000Z","isFree":false},{"id":"qwen/qwen-2.5-coder-32b-instruct","slug":"qwen-qwen-2-5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","shortName":"Qwen2.5 Coder 32B Instruct","providerId":"qwen","description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","contextLength":32768,"maxOutputTokens":29491,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.66,"outputPerMTok":1,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen-2.5-coder-32b-instruct"},"blendedPerMTok":0.745,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-06-30","openWeights":true,"huggingFaceId":"Qwen/Qwen2.5-Coder-32B-Instruct","benchmarks":null,"releasedAt":"2024-11-11T23:40:00.000Z","isFree":false},{"id":"thedrummer/unslopnemo-12b","slug":"thedrummer-unslopnemo-12b","name":"TheDrummer: UnslopNemo 12B","shortName":"UnslopNemo 12B","providerId":"thedrummer","description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","contextLength":1024000,"maxOutputTokens":819200,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.39999999999999997,"outputPerMTok":0.39999999999999997,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/thedrummer/unslopnemo-12b"},"blendedPerMTok":0.39999999999999997,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-04-30","openWeights":true,"huggingFaceId":"TheDrummer/UnslopNemo-12B-v4.1","benchmarks":null,"releasedAt":"2024-11-08T22:04:08.000Z","isFree":false},{"id":"anthracite-org/magnum-v4-72b","slug":"anthracite-org-magnum-v4-72b","name":"Magnum v4 72B","shortName":"Magnum v4 72B","providerId":"anthracite-org","description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","contextLength":32768,"maxOutputTokens":4096,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":2.5,"outputPerMTok":5,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthracite-org/magnum-v4-72b"},"blendedPerMTok":3.125,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-06-30","openWeights":true,"huggingFaceId":"anthracite-org/magnum-v4-72b","benchmarks":null,"releasedAt":"2024-10-22T00:00:00.000Z","isFree":false},{"id":"qwen/qwen-2.5-7b-instruct","slug":"qwen-qwen-2-5-7b-instruct","name":"Qwen: Qwen2.5 7B Instruct","shortName":"Qwen2.5 7B Instruct","providerId":"qwen","description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","contextLength":32768,"maxOutputTokens":29491,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.09999999999999999,"outputPerMTok":0.19999999999999998,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen-2.5-7b-instruct"},"blendedPerMTok":0.125,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-06-30","openWeights":true,"huggingFaceId":"Qwen/Qwen2.5-7B-Instruct","benchmarks":null,"releasedAt":"2024-10-16T00:00:00.000Z","isFree":false},{"id":"meta-llama/llama-3.2-1b-instruct","slug":"meta-llama-llama-3-2-1b-instruct","name":"Meta: Llama 3.2 1B Instruct","shortName":"Llama 3.2 1B Instruct","providerId":"meta-llama","description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","contextLength":60000,"maxOutputTokens":54000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.027,"outputPerMTok":0.201,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta-llama/llama-3.2-1b-instruct"},"blendedPerMTok":0.07050000000000001,"capabilities":{"tools":false,"structuredOutput":false,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":true,"huggingFaceId":"meta-llama/Llama-3.2-1B-Instruct","benchmarks":null,"releasedAt":"2024-09-25T00:00:00.000Z","isFree":false},{"id":"meta-llama/llama-3.2-3b-instruct","slug":"meta-llama-llama-3-2-3b-instruct","name":"Meta: Llama 3.2 3B Instruct","shortName":"Llama 3.2 3B Instruct","providerId":"meta-llama","description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.049999999999999996,"outputPerMTok":0.33,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta-llama/llama-3.2-3b-instruct"},"blendedPerMTok":0.12,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":true,"huggingFaceId":"meta-llama/Llama-3.2-3B-Instruct","benchmarks":null,"releasedAt":"2024-09-25T00:00:00.000Z","isFree":false},{"id":"qwen/qwen-2.5-72b-instruct","slug":"qwen-qwen-2-5-72b-instruct","name":"Qwen2.5 72B Instruct","shortName":"Qwen2.5 72B Instruct","providerId":"qwen","description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","contextLength":32768,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.36,"outputPerMTok":0.39999999999999997,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/qwen/qwen-2.5-72b-instruct"},"blendedPerMTok":0.37,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-06-30","openWeights":true,"huggingFaceId":"Qwen/Qwen2.5-72B-Instruct","benchmarks":null,"releasedAt":"2024-09-19T00:00:00.000Z","isFree":false},{"id":"cohere/command-r-08-2024","slug":"cohere-command-r-08-2024","name":"Cohere: Command R (08-2024)","shortName":"Command R (08-2024)","providerId":"cohere","description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","contextLength":128000,"maxOutputTokens":4000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":0.6,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/cohere/command-r-08-2024"},"blendedPerMTok":0.26249999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-03-31","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2024-08-30T00:00:00.000Z","isFree":false},{"id":"cohere/command-r-plus-08-2024","slug":"cohere-command-r-plus-08-2024","name":"Cohere: Command R+ (08-2024)","shortName":"Command R+ (08-2024)","providerId":"cohere","description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","contextLength":128000,"maxOutputTokens":4000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":2.5,"outputPerMTok":10,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/cohere/command-r-plus-08-2024"},"blendedPerMTok":4.375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-03-31","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2024-08-30T00:00:00.000Z","isFree":false},{"id":"sao10k/l3.1-euryale-70b","slug":"sao10k-l3-1-euryale-70b","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","shortName":"Llama 3.1 Euryale 70B v2.2","providerId":"sao10k","description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.85,"outputPerMTok":0.85,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/sao10k/l3.1-euryale-70b"},"blendedPerMTok":0.85,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":true,"huggingFaceId":"Sao10K/L3.1-70B-Euryale-v2.2","benchmarks":null,"releasedAt":"2024-08-28T00:00:00.000Z","isFree":false},{"id":"nousresearch/hermes-3-llama-3.1-70b","slug":"nousresearch-hermes-3-llama-3-1-70b","name":"Nous: Hermes 3 70B Instruct","shortName":"Hermes 3 70B Instruct","providerId":"nousresearch","description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.7,"outputPerMTok":0.7,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/nousresearch/hermes-3-llama-3.1-70b"},"blendedPerMTok":0.7,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":true,"huggingFaceId":"NousResearch/Hermes-3-Llama-3.1-70B","benchmarks":null,"releasedAt":"2024-08-18T00:00:00.000Z","isFree":false},{"id":"nousresearch/hermes-3-llama-3.1-405b","slug":"nousresearch-hermes-3-llama-3-1-405b","name":"Nous: Hermes 3 405B Instruct","shortName":"Hermes 3 405B Instruct","providerId":"nousresearch","description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":1,"outputPerMTok":1,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/nousresearch/hermes-3-llama-3.1-405b"},"blendedPerMTok":1,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":true,"huggingFaceId":"NousResearch/Hermes-3-Llama-3.1-405B","benchmarks":null,"releasedAt":"2024-08-16T00:00:00.000Z","isFree":false},{"id":"sao10k/l3-lunaris-8b","slug":"sao10k-l3-lunaris-8b","name":"Sao10K: Llama 3 8B Lunaris","shortName":"Llama 3 8B Lunaris","providerId":"sao10k","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","contextLength":8192,"maxOutputTokens":7372,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.04,"outputPerMTok":0.049999999999999996,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/sao10k/l3-lunaris-8b"},"blendedPerMTok":0.042499999999999996,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":true,"huggingFaceId":"Sao10K/L3-8B-Lunaris-v1","benchmarks":null,"releasedAt":"2024-08-13T00:00:00.000Z","isFree":false},{"id":"openai/gpt-4o-2024-08-06","slug":"openai-gpt-4o-2024-08-06","name":"OpenAI: GPT-4o (2024-08-06)","shortName":"GPT-4o (2024-08-06)","providerId":"openai","description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","contextLength":128000,"maxOutputTokens":16384,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":2.5,"outputPerMTok":10,"cachedInputPerMTok":1.25,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4o-2024-08-06"},"blendedPerMTok":4.375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2023-10-31","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2024-08-06T00:00:00.000Z","isFree":false},{"id":"meta-llama/llama-3.1-70b-instruct","slug":"meta-llama-llama-3-1-70b-instruct","name":"Meta: Llama 3.1 70B Instruct","shortName":"Llama 3.1 70B Instruct","providerId":"meta-llama","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.39999999999999997,"outputPerMTok":0.39999999999999997,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta-llama/llama-3.1-70b-instruct"},"blendedPerMTok":0.39999999999999997,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":true,"huggingFaceId":"meta-llama/Meta-Llama-3.1-70B-Instruct","benchmarks":null,"releasedAt":"2024-07-23T00:00:00.000Z","isFree":false},{"id":"meta-llama/llama-3.1-8b-instruct","slug":"meta-llama-llama-3-1-8b-instruct","name":"Meta: Llama 3.1 8B Instruct","shortName":"Llama 3.1 8B Instruct","providerId":"meta-llama","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","contextLength":131072,"maxOutputTokens":117964,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.049999999999999996,"outputPerMTok":0.08,"cachedInputPerMTok":0.024999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/meta-llama/llama-3.1-8b-instruct"},"blendedPerMTok":0.057499999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":true,"huggingFaceId":"meta-llama/Meta-Llama-3.1-8B-Instruct","benchmarks":{"coding_index":5.4},"releasedAt":"2024-07-23T00:00:00.000Z","isFree":false},{"id":"mistralai/mistral-nemo","slug":"mistralai-mistral-nemo","name":"Mistral: Mistral Nemo","shortName":"Mistral Nemo","providerId":"mistralai","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","contextLength":131072,"maxOutputTokens":16384,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.019000000000000003,"outputPerMTok":0.03,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-nemo"},"blendedPerMTok":0.021750000000000002,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-04-30","openWeights":true,"huggingFaceId":"mistralai/Mistral-Nemo-Instruct-2407","benchmarks":null,"releasedAt":"2024-07-19T00:00:00.000Z","isFree":false},{"id":"openai/gpt-4o-mini","slug":"openai-gpt-4o-mini","name":"OpenAI: GPT-4o-mini","shortName":"GPT-4o-mini","providerId":"openai","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","contextLength":128000,"maxOutputTokens":16384,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":0.6,"cachedInputPerMTok":0.075,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4o-mini"},"blendedPerMTok":0.26249999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2023-10-31","openWeights":false,"huggingFaceId":null,"benchmarks":{"coding_index":11.4},"releasedAt":"2024-07-18T00:00:00.000Z","isFree":false},{"id":"openai/gpt-4o-mini-2024-07-18","slug":"openai-gpt-4o-mini-2024-07-18","name":"OpenAI: GPT-4o-mini (2024-07-18)","shortName":"GPT-4o-mini (2024-07-18)","providerId":"openai","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","contextLength":128000,"maxOutputTokens":16384,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":0.15,"outputPerMTok":0.6,"cachedInputPerMTok":0.075,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4o-mini-2024-07-18"},"blendedPerMTok":0.26249999999999996,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2023-10-31","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2024-07-18T00:00:00.000Z","isFree":false},{"id":"google/gemma-2-27b-it","slug":"google-gemma-2-27b-it","name":"Google: Gemma 2 27B","shortName":"Gemma 2 27B","providerId":"google","description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","contextLength":8192,"maxOutputTokens":2048,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.65,"outputPerMTok":0.65,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/google/gemma-2-27b-it"},"blendedPerMTok":0.65,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-06-30","openWeights":true,"huggingFaceId":"google/gemma-2-27b-it","benchmarks":null,"releasedAt":"2024-07-13T00:00:00.000Z","isFree":false},{"id":"openai/gpt-4o","slug":"openai-gpt-4o","name":"OpenAI: GPT-4o","shortName":"GPT-4o","providerId":"openai","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","contextLength":128000,"maxOutputTokens":16384,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":2.5,"outputPerMTok":10,"cachedInputPerMTok":1.25,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4o"},"blendedPerMTok":4.375,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":true},"knowledgeCutoff":"2023-10-31","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2024-05-13T00:00:00.000Z","isFree":false},{"id":"openai/gpt-4o-2024-05-13","slug":"openai-gpt-4o-2024-05-13","name":"OpenAI: GPT-4o (2024-05-13)","shortName":"GPT-4o (2024-05-13)","providerId":"openai","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","contextLength":128000,"maxOutputTokens":4096,"inputModalities":["text","image","file"],"outputModalities":["text"],"price":{"inputPerMTok":5,"outputPerMTok":15,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4o-2024-05-13"},"blendedPerMTok":7.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-10-31","openWeights":false,"huggingFaceId":null,"benchmarks":{"coding_index":24.2},"releasedAt":"2024-05-13T00:00:00.000Z","isFree":false},{"id":"mistralai/mixtral-8x22b-instruct","slug":"mistralai-mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","shortName":"Mixtral 8x22B Instruct","providerId":"mistralai","description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","contextLength":65536,"maxOutputTokens":52428,"inputModalities":["text","file"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":6,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mixtral-8x22b-instruct"},"blendedPerMTok":3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-01-31","openWeights":true,"huggingFaceId":"mistralai/Mixtral-8x22B-Instruct-v0.1","benchmarks":null,"releasedAt":"2024-04-17T00:00:00.000Z","isFree":false},{"id":"microsoft/wizardlm-2-8x22b","slug":"microsoft-wizardlm-2-8x22b","name":"WizardLM-2 8x22B","shortName":"WizardLM-2 8x22B","providerId":"microsoft","description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","contextLength":65535,"maxOutputTokens":8000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.62,"outputPerMTok":0.62,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/microsoft/wizardlm-2-8x22b"},"blendedPerMTok":0.62,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2024-04-30","openWeights":true,"huggingFaceId":"microsoft/WizardLM-2-8x22B","benchmarks":null,"releasedAt":"2024-04-16T00:00:00.000Z","isFree":false},{"id":"openai/gpt-4-turbo","slug":"openai-gpt-4-turbo","name":"OpenAI: GPT-4 Turbo","shortName":"GPT-4 Turbo","providerId":"openai","description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","contextLength":128000,"maxOutputTokens":4096,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":10,"outputPerMTok":30,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4-turbo"},"blendedPerMTok":15,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":true},"knowledgeCutoff":"2023-12-31","openWeights":false,"huggingFaceId":null,"benchmarks":{"coding_index":21.5},"releasedAt":"2024-04-09T00:00:00.000Z","isFree":false},{"id":"anthropic/claude-3-haiku","slug":"anthropic-claude-3-haiku","name":"Anthropic: Claude 3 Haiku","shortName":"Claude 3 Haiku","providerId":"anthropic","description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","contextLength":200000,"maxOutputTokens":4096,"inputModalities":["text","image"],"outputModalities":["text"],"price":{"inputPerMTok":0.25,"outputPerMTok":1.25,"cachedInputPerMTok":0.03,"cacheWritePerMTok":0.3,"perRequest":null,"perImage":null,"searchPerRequest":0.01,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/anthropic/claude-3-haiku"},"blendedPerMTok":0.5,"capabilities":{"tools":true,"structuredOutput":false,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2023-08-31","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2024-03-13T00:00:00.000Z","isFree":false},{"id":"mistralai/mistral-large","slug":"mistralai-mistral-large","name":"Mistral Large","shortName":"Mistral Large","providerId":"mistralai","description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","contextLength":128000,"maxOutputTokens":102400,"inputModalities":["text","file"],"outputModalities":["text"],"price":{"inputPerMTok":2,"outputPerMTok":6,"cachedInputPerMTok":0.19999999999999998,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mistralai/mistral-large"},"blendedPerMTok":3,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":true,"streaming":true,"batch":false},"knowledgeCutoff":"2024-11-30","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2024-02-26T00:00:00.000Z","isFree":false},{"id":"openai/gpt-3.5-turbo-0613","slug":"openai-gpt-3-5-turbo-0613","name":"OpenAI: GPT-3.5 Turbo (older v0613)","shortName":"GPT-3.5 Turbo (older v0613)","providerId":"openai","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","contextLength":4095,"maxOutputTokens":3685,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":1,"outputPerMTok":2,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-3.5-turbo-0613"},"blendedPerMTok":1.25,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2021-09-30","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2024-01-25T00:00:00.000Z","isFree":false},{"id":"openai/gpt-4-turbo-preview","slug":"openai-gpt-4-turbo-preview","name":"OpenAI: GPT-4 Turbo Preview","shortName":"GPT-4 Turbo Preview","providerId":"openai","description":"The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...","contextLength":128000,"maxOutputTokens":4096,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":10,"outputPerMTok":30,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4-turbo-preview"},"blendedPerMTok":15,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-12-31","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2024-01-25T00:00:00.000Z","isFree":false},{"id":"openrouter/auto","slug":"openrouter-auto","name":"Auto Router","shortName":"Auto Router","providerId":"openrouter","description":"The Auto Router automatically selects the best model for your prompt, powered by the wisdom of the market. It routes you based on what the OpenRouter community collectively spends on...","contextLength":2000000,"maxOutputTokens":null,"inputModalities":["text","image","audio","file","video"],"outputModalities":["text","image"],"price":{"inputPerMTok":-1000000,"outputPerMTok":-1000000,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openrouter/auto"},"blendedPerMTok":-1000000,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":true,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":null,"openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2023-11-08T00:00:00.000Z","isFree":false},{"id":"openai/gpt-3.5-turbo-instruct","slug":"openai-gpt-3-5-turbo-instruct","name":"OpenAI: GPT-3.5 Turbo Instruct","shortName":"GPT-3.5 Turbo Instruct","providerId":"openai","description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","contextLength":4095,"maxOutputTokens":3685,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":1.5,"outputPerMTok":2,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-3.5-turbo-instruct"},"blendedPerMTok":1.625,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2021-09-30","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2023-09-28T00:00:00.000Z","isFree":false},{"id":"openai/gpt-3.5-turbo-16k","slug":"openai-gpt-3-5-turbo-16k","name":"OpenAI: GPT-3.5 Turbo 16k","shortName":"GPT-3.5 Turbo 16k","providerId":"openai","description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","contextLength":16385,"maxOutputTokens":4096,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":3,"outputPerMTok":4,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-3.5-turbo-16k"},"blendedPerMTok":3.25,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2021-09-30","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2023-08-28T00:00:00.000Z","isFree":false},{"id":"mancer/weaver","slug":"mancer-weaver","name":"Mancer: Weaver (alpha)","shortName":"Weaver (alpha)","providerId":"mancer","description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","contextLength":8000,"maxOutputTokens":6000,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.39999999999999997,"outputPerMTok":0.75,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/mancer/weaver"},"blendedPerMTok":0.4875,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-06-30","openWeights":false,"huggingFaceId":null,"benchmarks":null,"releasedAt":"2023-08-02T00:00:00.000Z","isFree":false},{"id":"undi95/remm-slerp-l2-13b","slug":"undi95-remm-slerp-l2-13b","name":"ReMM SLERP 13B","shortName":"ReMM SLERP 13B","providerId":"undi95","description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","contextLength":6144,"maxOutputTokens":5529,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.35,"outputPerMTok":0.65,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/undi95/remm-slerp-l2-13b"},"blendedPerMTok":0.42499999999999993,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-06-30","openWeights":true,"huggingFaceId":"Undi95/ReMM-SLERP-L2-13B","benchmarks":null,"releasedAt":"2023-07-22T00:00:00.000Z","isFree":false},{"id":"gryphe/mythomax-l2-13b","slug":"gryphe-mythomax-l2-13b","name":"MythoMax 13B","shortName":"MythoMax 13B","providerId":"gryphe","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","contextLength":8192,"maxOutputTokens":3686,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.06,"outputPerMTok":0.06,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/gryphe/mythomax-l2-13b"},"blendedPerMTok":0.06,"capabilities":{"tools":false,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2023-06-30","openWeights":true,"huggingFaceId":"Gryphe/MythoMax-L2-13b","benchmarks":null,"releasedAt":"2023-07-02T00:00:00.000Z","isFree":false},{"id":"openai/gpt-3.5-turbo","slug":"openai-gpt-3-5-turbo","name":"OpenAI: GPT-3.5 Turbo","shortName":"GPT-3.5 Turbo","providerId":"openai","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","contextLength":16385,"maxOutputTokens":4096,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":0.5,"outputPerMTok":1.5,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-3.5-turbo"},"blendedPerMTok":0.75,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":true},"knowledgeCutoff":"2021-09-30","openWeights":false,"huggingFaceId":null,"benchmarks":{"coding_index":10.7},"releasedAt":"2023-05-28T00:00:00.000Z","isFree":false},{"id":"openai/gpt-4","slug":"openai-gpt-4","name":"OpenAI: GPT-4","shortName":"GPT-4","providerId":"openai","description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","contextLength":8191,"maxOutputTokens":4096,"inputModalities":["text"],"outputModalities":["text"],"price":{"inputPerMTok":30,"outputPerMTok":60,"cachedInputPerMTok":null,"cacheWritePerMTok":null,"perRequest":null,"perImage":null,"searchPerRequest":null,"source":"openrouter","confirmedAt":"2026-09-14","sourceUrl":"https://openrouter.ai/openai/gpt-4"},"blendedPerMTok":37.5,"capabilities":{"tools":true,"structuredOutput":true,"reasoning":false,"promptCaching":false,"streaming":true,"batch":false},"knowledgeCutoff":"2021-09-30","openWeights":false,"huggingFaceId":null,"benchmarks":{"coding_index":13.1},"releasedAt":"2023-05-28T00:00:00.000Z","isFree":false}]}