{"stepfun/step-3.7-flash":{"maxTokens":256000,"contextWindow":256000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.19999999999999998,"outputPrice":1.15,"cacheReadsPrice":0.04,"description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"anthropic/claude-opus-4.8-fast":{"maxTokens":128000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":10,"outputPrice":50,"cacheWritesPrice":12.5,"cacheReadsPrice":1,"description":"Fast-mode variant of [Opus 4.8](/anthropic/claude-opus-4.8) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 4.8.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"anthropic/claude-opus-4.8":{"maxTokens":128000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":5,"outputPrice":25,"cacheWritesPrice":6.25,"cacheReadsPrice":0.5,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"qwen/qwen3.7-max":{"maxTokens":65536,"contextWindow":1000000,"supportsImages":false,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":3.75,"cacheWritesPrice":1.5625,"cacheReadsPrice":0.25,"description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"x-ai/grok-build-0.1":{"maxTokens":51200,"contextWindow":256000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1,"outputPrice":2,"cacheReadsPrice":0.19999999999999998,"description":"Grok Build 0.1 is xAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/gemini-3.5-flash":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.5,"outputPrice":9,"cacheWritesPrice":0.08333333333333334,"cacheReadsPrice":0.15,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"anthropic/claude-opus-4.7-fast":{"maxTokens":128000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":30,"outputPrice":150,"cacheWritesPrice":37.5,"cacheReadsPrice":3,"description":"Fast-mode variant of [Opus 4.7](/anthropic/claude-opus-4.7) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"perceptron/perceptron-mk1":{"maxTokens":8192,"contextWindow":32768,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.15,"outputPrice":1.5,"description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"inclusionai/ring-2.6-1t":{"maxTokens":65536,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.075,"outputPrice":0.625,"cacheReadsPrice":0.015,"description":"Ring-2.6-1T is a 1T-parameter-scale thinking model with 63B active parameters, built for real-world agent workflows that require both strong capability and operational efficiency. It is optimized for coding agents, tool...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/gemini-3.1-flash-lite":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.25,"outputPrice":1.5,"cacheWritesPrice":0.08333333333333334,"cacheReadsPrice":0.024999999999999998,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-chat-latest":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":5,"outputPrice":30,"cacheReadsPrice":0.5,"description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","supportsReasoningEffort":false,"supportedParameters":["max_tokens"]},"x-ai/grok-4.3":{"maxTokens":200000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":2.5,"cacheReadsPrice":0.19999999999999998,"description":"Grok 4.3 is a reasoning model from xAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"ibm-granite/granite-4.1-8b":{"maxTokens":131072,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.049999999999999996,"outputPrice":0.09999999999999999,"cacheReadsPrice":0.049999999999999996,"description":"Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mistralai/mistral-medium-3-5":{"maxTokens":52429,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":1.5,"outputPrice":7.5,"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openrouter/owl-alpha":{"maxTokens":262144,"contextWindow":1048756,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Owl Alpha is a high-performance foundation model designed for agentic workloads. Natively supports tool use, and long-context tasks, with strong performance in code generation, automated workflows, and complex instruction execution....","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free":{"maxTokens":65536,"contextWindow":256000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"NVIDIA Nemotron™ 3 Nano Omni is a 30B-A3B open multimodal model designed to function as a perception and context sub-agent in enterprise agent systems. It accepts text, image, video, and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"poolside/laguna-xs.2:free":{"maxTokens":32768,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Laguna XS.2 is the second-generation model in the XS size class from [Poolside](https://poolside.ai), their efficient coding agent series. It combines tool calling and reasoning capabilities with a compact footprint, offering...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"poolside/laguna-m.1:free":{"maxTokens":32768,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Laguna M.1 is the flagship coding agent model from [Poolside](https://poolside.ai), optimized for complex software engineering tasks. Designed for agentic coding workflows, it supports tool calling and reasoning, with a 128K...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"~anthropic/claude-haiku-latest":{"maxTokens":64000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1,"outputPrice":5,"cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999,"description":"This model always redirects to the latest model in the Anthropic Claude Haiku family.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"~openai/gpt-mini-latest":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.75,"outputPrice":4.5,"cacheReadsPrice":0.075,"description":"This model always redirects to the latest model in the OpenAI GPT Mini family.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"~google/gemini-pro-latest":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":2,"outputPrice":12,"cacheWritesPrice":0.375,"cacheReadsPrice":0.19999999999999998,"description":"This model always redirects to the latest model in the Google Gemini Pro family.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"~moonshotai/kimi-latest":{"maxTokens":262144,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.684,"outputPrice":3.42,"cacheReadsPrice":0.144,"description":"This model always redirects to the latest model in the MoonshotAI Kimi family.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"~google/gemini-flash-latest":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.5,"outputPrice":9,"cacheWritesPrice":0.08333333333333334,"cacheReadsPrice":0.15,"description":"This model always redirects to the latest model in the Google Gemini Flash family.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"~anthropic/claude-sonnet-latest":{"maxTokens":128000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":3,"outputPrice":15,"cacheWritesPrice":3.75,"cacheReadsPrice":0.3,"description":"This model always redirects to the latest model in the Anthropic Claude Sonnet family.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"~openai/gpt-latest":{"maxTokens":128000,"contextWindow":1050000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":5,"outputPrice":30,"cacheReadsPrice":0.5,"description":"This model always redirects to the latest model in the OpenAI GPT family.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"qwen/qwen3.5-plus-20260420":{"maxTokens":65536,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.3,"outputPrice":1.7999999999999998,"cacheWritesPrice":0.375,"description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3.6-flash":{"maxTokens":65536,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.1875,"outputPrice":1.125,"cacheWritesPrice":0.234375,"description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3.6-35b-a3b":{"maxTokens":262140,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.14,"outputPrice":1,"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3.6-max-preview":{"maxTokens":65536,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":1.04,"outputPrice":6.24,"cacheWritesPrice":1.3,"description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3.6-27b":{"maxTokens":262140,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.29,"outputPrice":3.1999999999999997,"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-5.5-pro":{"maxTokens":128000,"contextWindow":1050000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":30,"outputPrice":180,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/gpt-5.5":{"maxTokens":128000,"contextWindow":1050000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":5,"outputPrice":30,"cacheReadsPrice":0.5,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"deepseek/deepseek-v4-pro":{"maxTokens":384000,"contextWindow":1048576,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.435,"outputPrice":0.87,"cacheReadsPrice":0.003625,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"deepseek/deepseek-v4-flash:free":{"maxTokens":384000,"contextWindow":1048576,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","reasoning"]},"deepseek/deepseek-v4-flash":{"maxTokens":131072,"contextWindow":1048576,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.0983,"outputPrice":0.1966,"cacheReadsPrice":0.019700000000000002,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"inclusionai/ling-2.6-1t":{"maxTokens":32768,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.075,"outputPrice":0.625,"cacheReadsPrice":0.015,"description":"Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"tencent/hy3-preview":{"maxTokens":52429,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.063,"outputPrice":0.21,"cacheReadsPrice":0.020999999999999998,"description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"xiaomi/mimo-v2.5-pro":{"maxTokens":131072,"contextWindow":1048576,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.435,"outputPrice":0.87,"cacheReadsPrice":0.0036,"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"xiaomi/mimo-v2.5":{"maxTokens":131072,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.14,"outputPrice":0.28,"cacheReadsPrice":0.0028,"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"inclusionai/ling-2.6-flash":{"maxTokens":32768,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.01,"outputPrice":0.03,"cacheReadsPrice":0.002,"description":"Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"~anthropic/claude-opus-latest":{"maxTokens":128000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":5,"outputPrice":25,"cacheWritesPrice":6.25,"cacheReadsPrice":0.5,"description":"This model always redirects to the latest model in the Claude Opus family.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openrouter/pareto-code":{"maxTokens":400000,"contextWindow":2000000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":-1000000,"outputPrice":-1000000,"description":"The Pareto Router maintains a tiered shortlist of strong coding models, ranked by [Artificial Analysis](https://artificialanalysis.ai/) coding percentiles. Set min_coding_score between 0 and 1 on the [pareto-router plugin](https://openrouter.ai/docs/guides/routing/routers/pareto-router#the-min_coding_score-parameter) to control how...","supportsReasoningEffort":false,"supportedParameters":[]},"moonshotai/kimi-k2.6:free":{"maxTokens":52429,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","reasoning"]},"moonshotai/kimi-k2.6":{"maxTokens":262144,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.684,"outputPrice":3.42,"cacheReadsPrice":0.144,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"anthropic/claude-opus-4.7":{"maxTokens":128000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":5,"outputPrice":25,"cacheWritesPrice":6.25,"cacheReadsPrice":0.5,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"anthropic/claude-opus-4.6-fast":{"maxTokens":128000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":30,"outputPrice":150,"cacheWritesPrice":37.5,"cacheReadsPrice":3,"description":"Fast-mode variant of [Opus 4.6](/anthropic/claude-opus-4.6) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"z-ai/glm-5.1":{"maxTokens":40551,"contextWindow":202752,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.98,"outputPrice":3.08,"cacheReadsPrice":0.182,"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/gemma-4-26b-a4b-it:free":{"maxTokens":32768,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/gemma-4-26b-a4b-it":{"maxTokens":52429,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.06,"outputPrice":0.33,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/gemma-4-31b-it:free":{"maxTokens":32768,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/gemma-4-31b-it":{"maxTokens":16384,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.12,"outputPrice":0.37,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3.6-plus":{"maxTokens":65536,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.325,"outputPrice":1.95,"cacheWritesPrice":0.40625,"description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"z-ai/glm-5v-turbo":{"maxTokens":131072,"contextWindow":202752,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.2,"outputPrice":4,"cacheReadsPrice":0.24,"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"arcee-ai/trinity-large-thinking":{"maxTokens":262144,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.22,"outputPrice":0.85,"cacheReadsPrice":0.06,"description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"x-ai/grok-4.20-multi-agent":{"maxTokens":400000,"contextWindow":2000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":2,"outputPrice":6,"cacheReadsPrice":0.19999999999999998,"description":"Grok 4.20 Multi-Agent is a variant of xAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"x-ai/grok-4.20":{"maxTokens":400000,"contextWindow":2000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":2.5,"cacheReadsPrice":0.19999999999999998,"description":"Grok 4.20 is a reasoning model from xAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/lyria-3-pro-preview":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Full-length songs are priced at $0.08 per song. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate high-quality, 48kHz...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"google/lyria-3-clip-preview":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"kwaipilot/kat-coder-pro-v2":{"maxTokens":80000,"contextWindow":256000,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.3,"outputPrice":1.2,"cacheReadsPrice":0.06,"description":"KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"rekaai/reka-edge":{"maxTokens":16384,"contextWindow":16384,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.09999999999999999,"outputPrice":0.09999999999999999,"description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"xiaomi/mimo-v2-omni":{"maxTokens":65536,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.39999999999999997,"outputPrice":2,"cacheReadsPrice":0.08,"description":"MiMo-V2-Omni is a frontier omni-modal model that natively processes image, video, and audio inputs within a unified architecture. It combines strong multimodal perception with agentic capability - visual grounding, multi-step...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"xiaomi/mimo-v2-pro":{"maxTokens":131072,"contextWindow":1048576,"supportsImages":false,"supportsPromptCache":true,"inputPrice":1,"outputPrice":3,"cacheReadsPrice":0.19999999999999998,"description":"MiMo-V2-Pro is Xiaomi's flagship foundation model, featuring over 1T total parameters and a 1M context length, deeply optimized for agentic scenarios. It is highly adaptable to general agent frameworks like...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"minimax/minimax-m2.7":{"maxTokens":40960,"contextWindow":204800,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.26,"outputPrice":1.2,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-5.4-nano":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.19999999999999998,"outputPrice":1.25,"cacheReadsPrice":0.02,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/gpt-5.4-mini":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.75,"outputPrice":4.5,"cacheReadsPrice":0.075,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"mistralai/mistral-small-2603":{"maxTokens":52429,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.15,"outputPrice":0.6,"cacheReadsPrice":0.015,"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"z-ai/glm-5-turbo":{"maxTokens":131072,"contextWindow":202752,"supportsImages":false,"supportsPromptCache":true,"inputPrice":1.2,"outputPrice":4,"cacheReadsPrice":0.24,"description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"nvidia/nemotron-3-super-120b-a12b:free":{"maxTokens":262144,"contextWindow":1000000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"nvidia/nemotron-3-super-120b-a12b":{"maxTokens":200000,"contextWindow":1000000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.09,"outputPrice":0.44999999999999996,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"bytedance-seed/seed-2.0-lite":{"maxTokens":131072,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.25,"outputPrice":2,"description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3.5-9b":{"maxTokens":81920,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.04,"outputPrice":0.15,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-5.4-pro":{"maxTokens":128000,"contextWindow":1050000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":30,"outputPrice":180,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/gpt-5.4":{"maxTokens":128000,"contextWindow":1050000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":2.5,"outputPrice":15,"cacheReadsPrice":0.25,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"inception/mercury-2":{"maxTokens":50000,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.25,"outputPrice":0.75,"cacheReadsPrice":0.024999999999999998,"description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-5.3-chat":{"maxTokens":16384,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.75,"outputPrice":14,"cacheReadsPrice":0.175,"description":"GPT-5.3 Chat is an update to ChatGPT's most-used model that makes everyday conversations smoother, more useful, and more directly helpful. It delivers more accurate answers with better contextualization and significantly...","supportsReasoningEffort":false,"supportedParameters":["max_tokens"]},"google/gemini-3.1-flash-lite-preview":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.25,"outputPrice":1.5,"cacheWritesPrice":0.08333333333333334,"cacheReadsPrice":0.024999999999999998,"description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"bytedance-seed/seed-2.0-mini":{"maxTokens":131072,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3.5-35b-a3b":{"maxTokens":262144,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.14,"outputPrice":1,"cacheReadsPrice":0.049999999999999996,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3.5-27b":{"maxTokens":65536,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.195,"outputPrice":1.56,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3.5-122b-a10b":{"maxTokens":262144,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.26,"outputPrice":2.08,"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3.5-flash-02-23":{"maxTokens":65536,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.065,"outputPrice":0.26,"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"liquid/lfm-2-24b-a2b":{"maxTokens":25600,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.03,"outputPrice":0.12,"description":"LFM2-24B-A2B is the largest model in the LFM2 family of hybrid architectures designed for efficient on-device deployment. Built as a 24B parameter Mixture-of-Experts model with only 2B active parameters per...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"google/gemini-3.1-pro-preview-customtools":{"maxTokens":65536,"contextWindow":1048756,"supportsImages":true,"supportsPromptCache":true,"inputPrice":2,"outputPrice":12,"cacheWritesPrice":0.375,"cacheReadsPrice":0.19999999999999998,"description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-5.3-codex":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.75,"outputPrice":14,"cacheReadsPrice":0.175,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"aion-labs/aion-2.0":{"maxTokens":32768,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.7999999999999999,"outputPrice":1.5999999999999999,"cacheReadsPrice":0.19999999999999998,"description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/gemini-3.1-pro-preview":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":2,"outputPrice":12,"cacheWritesPrice":0.375,"cacheReadsPrice":0.19999999999999998,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"anthropic/claude-sonnet-4.6":{"maxTokens":64000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":3,"outputPrice":15,"cacheWritesPrice":3.75,"cacheReadsPrice":0.3,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"],"supportsReasoningBudget":true},"qwen/qwen3.5-plus-02-15":{"maxTokens":65536,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.26,"outputPrice":1.56,"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3.5-397b-a17b":{"maxTokens":65536,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.39,"outputPrice":2.34,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"minimax/minimax-m2.5":{"maxTokens":196608,"contextWindow":204800,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.15,"outputPrice":1.15,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"z-ai/glm-5":{"maxTokens":40551,"contextWindow":202752,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.6,"outputPrice":1.92,"cacheReadsPrice":0.12,"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-max-thinking":{"maxTokens":32768,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.78,"outputPrice":3.9,"description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"anthropic/claude-opus-4.6":{"maxTokens":128000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":5,"outputPrice":25,"cacheWritesPrice":6.25,"cacheReadsPrice":0.5,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"],"supportsReasoningBudget":true},"qwen/qwen3-coder-next":{"maxTokens":262144,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.11,"outputPrice":0.7999999999999999,"cacheReadsPrice":0.07,"description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openrouter/free":{"maxTokens":40000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"The simplest way to get free inference. openrouter/free is a router that selects free models at random from the models available on OpenRouter. The router smartly filters for models that...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"stepfun/step-3.5-flash":{"maxTokens":16384,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.09,"outputPrice":0.3,"cacheReadsPrice":0.02,"description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"moonshotai/kimi-k2.5":{"maxTokens":262144,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.39999999999999997,"outputPrice":1.9,"cacheReadsPrice":0.09,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"minimax/minimax-m2-her":{"maxTokens":2048,"contextWindow":65536,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.3,"outputPrice":1.2,"cacheReadsPrice":0.03,"description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"writer/palmyra-x5":{"maxTokens":8192,"contextWindow":1040000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.6,"outputPrice":6,"description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"liquid/lfm-2.5-1.2b-thinking:free":{"maxTokens":6554,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"LFM2.5-1.2B-Thinking is a lightweight reasoning-focused model optimized for agentic tasks, data extraction, and RAG—while still running comfortably on edge devices. It supports long context (up to 32K tokens) and is...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"liquid/lfm-2.5-1.2b-instruct:free":{"maxTokens":6554,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"LFM2.5-1.2B-Instruct is a compact, high-performance instruction-tuned model built for fast on-device AI. It delivers strong chat quality in a 1.2B parameter footprint, with efficient edge inference and broad runtime support.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-audio":{"maxTokens":16384,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":2.5,"outputPrice":10,"description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-audio-mini":{"maxTokens":16384,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.6,"outputPrice":2.4,"description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"z-ai/glm-4.7-flash":{"maxTokens":16384,"contextWindow":202752,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.06,"outputPrice":0.39999999999999997,"cacheReadsPrice":0.01,"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-5.2-codex":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.75,"outputPrice":14,"cacheReadsPrice":0.175,"description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"bytedance-seed/seed-1.6-flash":{"maxTokens":32768,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.075,"outputPrice":0.3,"description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"bytedance-seed/seed-1.6":{"maxTokens":32768,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.25,"outputPrice":2,"description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"minimax/minimax-m2.1":{"maxTokens":196608,"contextWindow":204800,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.29,"outputPrice":0.95,"cacheReadsPrice":0.03,"description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"z-ai/glm-4.7":{"maxTokens":131072,"contextWindow":202752,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.39999999999999997,"outputPrice":1.75,"cacheReadsPrice":0.08,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/gemini-3-flash-preview":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.5,"outputPrice":3,"cacheWritesPrice":0.08333333333333334,"cacheReadsPrice":0.049999999999999996,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"xiaomi/mimo-v2-flash":{"maxTokens":65536,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.09999999999999999,"outputPrice":0.3,"cacheReadsPrice":0.01,"description":"MiMo-V2-Flash is an open-source foundation language model developed by Xiaomi. It is a Mixture-of-Experts model with 309B total parameters and 15B active parameters, adopting hybrid attention architecture. MiMo-V2-Flash supports a...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"nvidia/nemotron-3-nano-30b-a3b:free":{"maxTokens":51200,"contextWindow":256000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"nvidia/nemotron-3-nano-30b-a3b":{"maxTokens":228000,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.049999999999999996,"outputPrice":0.19999999999999998,"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-5.2-chat":{"maxTokens":16384,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.75,"outputPrice":14,"cacheReadsPrice":0.175,"description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","supportsReasoningEffort":false,"supportedParameters":["max_tokens"]},"openai/gpt-5.2-pro":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":21,"outputPrice":168,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/gpt-5.2":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.75,"outputPrice":14,"cacheReadsPrice":0.175,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"mistralai/devstral-2512":{"maxTokens":52429,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.39999999999999997,"outputPrice":2,"cacheReadsPrice":0.04,"description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"relace/relace-search":{"maxTokens":128000,"contextWindow":256000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":1,"outputPrice":3,"description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"z-ai/glm-4.6v":{"maxTokens":24000,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.3,"outputPrice":0.8999999999999999,"cacheReadsPrice":0.049999999999999996,"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"nex-agi/deepseek-v3.1-nex-n1":{"maxTokens":163840,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.135,"outputPrice":0.5,"description":"DeepSeek V3.1 Nex-N1 is the flagship release of the Nex-N1 series — a post-trained model designed to highlight agent autonomy, tool use, and real-world productivity. Nex-N1 demonstrates competitive performance across...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"essentialai/rnj-1-instruct":{"maxTokens":6554,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.15,"outputPrice":0.15,"description":"Rnj-1 is an 8B-parameter, dense, open-weight model family developed by Essential AI and trained from scratch with a focus on programming, math, and scientific reasoning. The model demonstrates strong performance...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openrouter/bodybuilder":{"maxTokens":25600,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":-1000000,"outputPrice":-1000000,"description":"Transform your natural language requests into structured OpenRouter API request objects. Describe what you want to accomplish with AI models, and Body Builder will construct the appropriate API calls. Example:...","supportsReasoningEffort":false,"supportedParameters":[]},"openai/gpt-5.1-codex-max":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":10,"cacheReadsPrice":0.125,"description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"amazon/nova-2-lite-v1":{"maxTokens":65535,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"mistralai/ministral-14b-2512":{"maxTokens":52429,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.19999999999999998,"outputPrice":0.19999999999999998,"cacheReadsPrice":0.02,"description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mistralai/ministral-8b-2512":{"maxTokens":52429,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.15,"outputPrice":0.15,"cacheReadsPrice":0.015,"description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mistralai/ministral-3b-2512":{"maxTokens":26215,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.09999999999999999,"outputPrice":0.09999999999999999,"cacheReadsPrice":0.01,"description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mistralai/mistral-large-2512":{"maxTokens":52429,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.5,"outputPrice":1.5,"cacheReadsPrice":0.049999999999999996,"description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"arcee-ai/trinity-mini":{"maxTokens":131072,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.045,"outputPrice":0.15,"description":"Trinity Mini is a 26B-parameter (3B active) sparse mixture-of-experts language model featuring 128 experts with 8 active per token. Engineered for efficient reasoning over long contexts (131k) with robust function...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"deepseek/deepseek-v3.2":{"maxTokens":65536,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.252,"outputPrice":0.378,"cacheReadsPrice":0.0252,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"prime-intellect/intellect-3":{"maxTokens":131072,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.19999999999999998,"outputPrice":1.1,"description":"INTELLECT-3 is a 106B-parameter Mixture-of-Experts model (12B active) post-trained from GLM-4.5-Air-Base using supervised fine-tuning (SFT) followed by large-scale reinforcement learning (RL). It offers state-of-the-art performance for its size across math,...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"anthropic/claude-opus-4.5":{"maxTokens":32000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":5,"outputPrice":25,"cacheWritesPrice":6.25,"cacheReadsPrice":0.5,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"],"supportsReasoningBudget":true},"allenai/olmo-3-32b-think":{"maxTokens":65536,"contextWindow":65536,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.15,"outputPrice":0.5,"description":"Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"deepcogito/cogito-v2.1-671b":{"maxTokens":25600,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":1.25,"outputPrice":1.25,"description":"Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-5.1":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":10,"cacheReadsPrice":0.13,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/gpt-5.1-chat":{"maxTokens":32000,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":10,"cacheReadsPrice":0.13,"description":"GPT-5.1 Chat (AKA Instant is the fast, lightweight member of the 5.1 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","supportsReasoningEffort":false,"supportedParameters":["max_tokens"]},"openai/gpt-5.1-codex":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":10,"cacheReadsPrice":0.13,"description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/gpt-5.1-codex-mini":{"maxTokens":100000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.25,"outputPrice":2,"cacheReadsPrice":0.024999999999999998,"description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"moonshotai/kimi-k2-thinking":{"maxTokens":262144,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.6,"outputPrice":2.5,"description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"amazon/nova-premier-v1":{"maxTokens":32000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":2.5,"outputPrice":12.5,"cacheReadsPrice":0.625,"description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"perplexity/sonar-pro-search":{"maxTokens":8000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":3,"outputPrice":15,"description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"mistralai/voxtral-small-24b-2507":{"maxTokens":6400,"contextWindow":32000,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.09999999999999999,"outputPrice":0.3,"cacheReadsPrice":0.01,"description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-oss-safeguard-20b":{"maxTokens":65536,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.075,"outputPrice":0.3,"cacheReadsPrice":0.037,"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"nvidia/nemotron-nano-12b-v2-vl:free":{"maxTokens":128000,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"NVIDIA Nemotron Nano 2 VL is a 12-billion-parameter open multimodal reasoning model designed for video understanding and document intelligence. It introduces a hybrid Transformer-Mamba architecture, combining transformer-level accuracy with Mamba’s...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"minimax/minimax-m2":{"maxTokens":196608,"contextWindow":204800,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.255,"outputPrice":1,"cacheReadsPrice":0.03,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-vl-32b-instruct":{"maxTokens":32768,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.10400000000000001,"outputPrice":0.41600000000000004,"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"ibm-granite/granite-4.0-h-micro":{"maxTokens":131000,"contextWindow":131000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.017,"outputPrice":0.112,"description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"microsoft/phi-4-mini-instruct":{"maxTokens":128000,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.08,"outputPrice":0.35,"cacheReadsPrice":0.08,"description":"Phi-4-mini-instruct is a lightweight open model built upon synthetic data and filtered publicly available websites - with a focus on high-quality, reasoning dense data. The model belongs to the Phi-4...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"anthropic/claude-haiku-4.5":{"maxTokens":64000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1,"outputPrice":5,"cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","supportsReasoningEffort":false,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"],"supportsReasoningBudget":true},"qwen/qwen3-vl-8b-thinking":{"maxTokens":32768,"contextWindow":256000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.117,"outputPrice":1.365,"description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-vl-8b-instruct":{"maxTokens":32768,"contextWindow":256000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.08,"outputPrice":0.5,"description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/o3-deep-research":{"maxTokens":100000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":10,"outputPrice":40,"cacheReadsPrice":2.5,"description":"o3-deep-research is OpenAI's advanced model for deep research, designed to tackle complex, multi-step research tasks.\n\nNote: This model always uses the 'web_search' tool which adds additional cost.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/o4-mini-deep-research":{"maxTokens":100000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":2,"outputPrice":8,"cacheReadsPrice":0.5,"description":"o4-mini-deep-research is OpenAI's faster, more affordable deep research model—ideal for tackling complex, multi-step research tasks.\n\nNote: This model always uses the 'web_search' tool which adds additional cost.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"nvidia/llama-3.3-nemotron-super-49b-v1.5":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"description":"Llama-3.3-Nemotron-Super-49B-v1.5 is a 49B-parameter, English-centric reasoning/chat model derived from Meta’s Llama-3.3-70B-Instruct with a 128K context. It’s post-trained for agentic workflows (RAG, tool calling) via SFT across math, code, science, and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-vl-30b-a3b-thinking":{"maxTokens":32768,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.13,"outputPrice":1.56,"description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-vl-30b-a3b-instruct":{"maxTokens":32768,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.13,"outputPrice":0.52,"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-5-pro":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":15,"outputPrice":120,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"z-ai/glm-4.6":{"maxTokens":131072,"contextWindow":202752,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.43,"outputPrice":1.74,"cacheReadsPrice":0.08,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"anthropic/claude-sonnet-4.5":{"maxTokens":64000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":3,"outputPrice":15,"cacheWritesPrice":3.75,"cacheReadsPrice":0.3,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"],"supportsReasoningBudget":true},"deepseek/deepseek-v3.2-exp":{"maxTokens":65536,"contextWindow":163840,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.27,"outputPrice":0.41,"description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"thedrummer/cydonia-24b-v4.1":{"maxTokens":131072,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.3,"outputPrice":0.5,"cacheReadsPrice":0.15,"description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"relace/relace-apply-3":{"maxTokens":128000,"contextWindow":256000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.85,"outputPrice":1.25,"description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","supportsReasoningEffort":false,"supportedParameters":["max_tokens"]},"google/gemini-2.5-flash-lite-preview-09-2025":{"maxTokens":65535,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"cacheWritesPrice":0.08333333333333334,"cacheReadsPrice":0.01,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-vl-235b-a22b-thinking":{"maxTokens":32768,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.26,"outputPrice":2.6,"description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-vl-235b-a22b-instruct":{"maxTokens":16384,"contextWindow":262144,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.19999999999999998,"outputPrice":0.88,"cacheReadsPrice":0.11,"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen3-max":{"maxTokens":32768,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.78,"outputPrice":3.9,"cacheWritesPrice":0.975,"cacheReadsPrice":0.156,"description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen3-coder-plus":{"maxTokens":65536,"contextWindow":1000000,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.65,"outputPrice":3.25,"cacheWritesPrice":0.8125,"cacheReadsPrice":0.13,"description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-5-codex":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":10,"cacheReadsPrice":0.125,"description":"GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"deepseek/deepseek-v3.1-terminus":{"maxTokens":32768,"contextWindow":163840,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.27,"outputPrice":0.95,"cacheReadsPrice":0.13,"description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-coder-flash":{"maxTokens":65536,"contextWindow":1000000,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.195,"outputPrice":0.975,"cacheWritesPrice":0.24375,"cacheReadsPrice":0.039,"description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen3-next-80b-a3b-thinking":{"maxTokens":32768,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.0975,"outputPrice":0.78,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-next-80b-a3b-instruct:free":{"maxTokens":52429,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen3-next-80b-a3b-instruct":{"maxTokens":16384,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.09,"outputPrice":1.1,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen-plus-2025-07-28:thinking":{"maxTokens":32768,"contextWindow":1000000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.26,"outputPrice":0.78,"cacheWritesPrice":0.325,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen-plus-2025-07-28":{"maxTokens":32768,"contextWindow":1000000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.26,"outputPrice":0.78,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"nvidia/nemotron-nano-9b-v2:free":{"maxTokens":25600,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"NVIDIA-Nemotron-Nano-9B-v2 is a large language model (LLM) trained from scratch by NVIDIA, and designed as a unified model for both reasoning and non-reasoning tasks. It responds to user queries and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"nvidia/nemotron-nano-9b-v2":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.04,"outputPrice":0.16,"description":"NVIDIA-Nemotron-Nano-9B-v2 is a large language model (LLM) trained from scratch by NVIDIA, and designed as a unified model for both reasoning and non-reasoning tasks. It responds to user queries and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"moonshotai/kimi-k2-0905":{"maxTokens":262144,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.6,"outputPrice":2.5,"description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen3-30b-a3b-thinking-2507":{"maxTokens":131072,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.08,"outputPrice":0.39999999999999997,"cacheReadsPrice":0.08,"description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"nousresearch/hermes-4-70b":{"maxTokens":26215,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.13,"outputPrice":0.39999999999999997,"description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"nousresearch/hermes-4-405b":{"maxTokens":26215,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":1,"outputPrice":3,"description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"deepseek/deepseek-chat-v3.1":{"maxTokens":32768,"contextWindow":163840,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.21,"outputPrice":0.7899999999999999,"cacheReadsPrice":0.13,"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"mistralai/mistral-medium-3.1":{"maxTokens":26215,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.39999999999999997,"outputPrice":2,"cacheReadsPrice":0.04,"description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"baidu/ernie-4.5-vl-28b-a3b":{"maxTokens":8000,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.14,"outputPrice":0.56,"description":"A powerful multimodal Mixture-of-Experts chat model featuring 28B total parameters with 3B activated per token, delivering exceptional text and vision understanding through its innovative heterogeneous MoE structure with modality-isolated routing....","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"z-ai/glm-4.5v":{"maxTokens":16384,"contextWindow":65536,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.6,"outputPrice":1.7999999999999998,"cacheReadsPrice":0.11,"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"ai21/jamba-large-1.7":{"maxTokens":4096,"contextWindow":256000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":2,"outputPrice":8,"description":"Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-5-chat":{"maxTokens":16384,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":10,"cacheReadsPrice":0.125,"description":"GPT-5 Chat is designed for advanced, natural, multimodal, and context-aware conversations for enterprise applications.","supportsReasoningEffort":false,"supportedParameters":["max_tokens"]},"openai/gpt-5":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":10,"cacheReadsPrice":0.125,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/gpt-5-mini":{"maxTokens":128000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.25,"outputPrice":2,"cacheReadsPrice":0.024999999999999998,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/gpt-5-nano":{"maxTokens":80000,"contextWindow":400000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.049999999999999996,"outputPrice":0.39999999999999997,"cacheReadsPrice":0.01,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/gpt-oss-120b:free":{"maxTokens":131072,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-oss-120b":{"maxTokens":26215,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.039,"outputPrice":0.18,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-oss-20b:free":{"maxTokens":8192,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/gpt-oss-20b":{"maxTokens":26215,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.029,"outputPrice":0.14,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"anthropic/claude-opus-4.1":{"maxTokens":32000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":15,"outputPrice":75,"cacheWritesPrice":18.75,"cacheReadsPrice":1.5,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"],"supportsReasoningBudget":true},"mistralai/codestral-2508":{"maxTokens":51200,"contextWindow":256000,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.3,"outputPrice":0.8999999999999999,"cacheReadsPrice":0.03,"description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen3-coder-30b-a3b-instruct":{"maxTokens":32768,"contextWindow":160000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.07,"outputPrice":0.27,"description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen3-30b-a3b-instruct-2507":{"maxTokens":262144,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.09,"outputPrice":0.3,"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"z-ai/glm-4.5":{"maxTokens":98304,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.6,"outputPrice":2.2,"cacheReadsPrice":0.11,"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"z-ai/glm-4.5-air:free":{"maxTokens":96000,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"z-ai/glm-4.5-air":{"maxTokens":131070,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.125,"outputPrice":0.85,"cacheReadsPrice":0.06,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-235b-a22b-thinking-2507":{"maxTokens":262144,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.09999999999999999,"outputPrice":0.09999999999999999,"cacheReadsPrice":0.09999999999999999,"description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"z-ai/glm-4-32b":{"maxTokens":25600,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.09999999999999999,"outputPrice":0.09999999999999999,"description":"GLM 4 32B is a cost-effective foundation language model. It can efficiently perform complex tasks and has significantly enhanced capabilities in tool use, online search, and code-related intelligent tasks. It...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen3-coder:free":{"maxTokens":262000,"contextWindow":1048576,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen3-coder":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.22,"outputPrice":1.7999999999999998,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"bytedance/ui-tars-1.5-7b":{"maxTokens":2048,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.09999999999999999,"outputPrice":0.19999999999999998,"cacheReadsPrice":0.09999999999999999,"description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"google/gemini-2.5-flash-lite":{"maxTokens":65535,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"cacheWritesPrice":0.08333333333333334,"cacheReadsPrice":0.01,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-235b-a22b-2507":{"maxTokens":16384,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.071,"outputPrice":0.09999999999999999,"description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"switchpoint/router":{"maxTokens":26215,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.85,"outputPrice":3.4,"description":"Switchpoint AI's router instantly analyzes your request and directs it to the optimal AI from an ever-evolving library. As the world of LLMs advances, our router gets smarter, ensuring you...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"moonshotai/kimi-k2":{"maxTokens":32768,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.5700000000000001,"outputPrice":2.3,"description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"cognitivecomputations/dolphin-mistral-24b-venice-edition:free":{"maxTokens":6554,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"tencent/hunyuan-a13b-instruct":{"maxTokens":131072,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.14,"outputPrice":0.5700000000000001,"description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"morph/morph-v3-large":{"maxTokens":131072,"contextWindow":262144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.8999999999999999,"outputPrice":1.9,"description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"morph/morph-v3-fast":{"maxTokens":38000,"contextWindow":81920,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.7999999999999999,"outputPrice":1.2,"description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"baidu/ernie-4.5-vl-424b-a47b":{"maxTokens":16000,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.42,"outputPrice":1.25,"description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"baidu/ernie-4.5-300b-a47b":{"maxTokens":12000,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.28,"outputPrice":1.1,"description":"ERNIE-4.5-300B-A47B is a 300B parameter Mixture-of-Experts (MoE) language model developed by Baidu as part of the ERNIE 4.5 series. It activates 47B parameters per token and supports text generation in...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mistralai/mistral-small-3.2-24b-instruct":{"maxTokens":16384,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.075,"outputPrice":0.19999999999999998,"description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"minimax/minimax-m1":{"maxTokens":40000,"contextWindow":1000000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.39999999999999997,"outputPrice":2.2,"description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/gemini-2.5-flash":{"maxTokens":65535,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.3,"outputPrice":2.5,"cacheWritesPrice":0.08333333333333334,"cacheReadsPrice":0.03,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"],"supportsReasoningBudget":true},"google/gemini-2.5-pro":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":10,"cacheWritesPrice":0.375,"cacheReadsPrice":0.125,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"],"supportsReasoningBudget":true,"requiredReasoningBudget":true},"openai/o3-pro":{"maxTokens":100000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":20,"outputPrice":80,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"google/gemini-2.5-pro-preview":{"maxTokens":65536,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":10,"cacheWritesPrice":0.375,"cacheReadsPrice":0.125,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"],"supportsReasoningBudget":true},"deepseek/deepseek-r1-0528":{"maxTokens":32768,"contextWindow":163840,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.5,"outputPrice":2.1500000000000004,"cacheReadsPrice":0.35,"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"anthropic/claude-opus-4":{"maxTokens":32000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":15,"outputPrice":75,"cacheWritesPrice":18.75,"cacheReadsPrice":1.5,"description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"],"supportsReasoningBudget":true},"anthropic/claude-sonnet-4":{"maxTokens":64000,"contextWindow":1000000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":3,"outputPrice":15,"cacheWritesPrice":3.75,"cacheReadsPrice":0.3,"description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"],"supportsReasoningBudget":true},"google/gemma-3n-e4b-it":{"maxTokens":6554,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.06,"outputPrice":0.12,"description":"Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mistralai/mistral-medium-3":{"maxTokens":26215,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.39999999999999997,"outputPrice":2,"cacheReadsPrice":0.04,"description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"google/gemini-2.5-pro-preview-05-06":{"maxTokens":65535,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.25,"outputPrice":10,"cacheWritesPrice":0.375,"cacheReadsPrice":0.125,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"arcee-ai/spotlight":{"maxTokens":65537,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.18,"outputPrice":0.18,"description":"Spotlight is a 7‑billion‑parameter vision‑language model derived from Qwen 2.5‑VL and fine‑tuned by Arcee AI for tight image‑text grounding tasks. It offers a 32 k‑token context window, enabling rich multimodal...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"arcee-ai/maestro-reasoning":{"maxTokens":32000,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.8999999999999999,"outputPrice":3.3000000000000003,"description":"Maestro Reasoning is Arcee's flagship analysis model: a 32 B‑parameter derivative of Qwen 2.5‑32 B tuned with DPO and chain‑of‑thought RL for step‑by‑step logic. Compared to the earlier 7 B...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"arcee-ai/virtuoso-large":{"maxTokens":64000,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.75,"outputPrice":1.2,"description":"Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"arcee-ai/coder-large":{"maxTokens":6554,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.5,"outputPrice":0.7999999999999999,"description":"Coder‑Large is a 32 B‑parameter offspring of Qwen 2.5‑Instruct that has been further trained on permissively‑licensed GitHub, CodeSearchNet and synthetic bug‑fix corpora. It supports a 32k context window, enabling multi‑file...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-guard-4-12b":{"maxTokens":16384,"contextWindow":163840,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.18,"outputPrice":0.18,"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen3-30b-a3b":{"maxTokens":20000,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.09,"outputPrice":0.44999999999999996,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-8b":{"maxTokens":8192,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.049999999999999996,"outputPrice":0.39999999999999997,"cacheReadsPrice":0.049999999999999996,"description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-14b":{"maxTokens":40960,"contextWindow":131702,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.09999999999999999,"outputPrice":0.24,"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-32b":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.08,"outputPrice":0.28,"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"qwen/qwen3-235b-a22b":{"maxTokens":8192,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.45499999999999996,"outputPrice":1.8199999999999998,"description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"openai/o4-mini-high":{"maxTokens":100000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.1,"outputPrice":4.4,"cacheReadsPrice":0.275,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/o3":{"maxTokens":100000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":2,"outputPrice":8,"cacheReadsPrice":0.5,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/o4-mini":{"maxTokens":100000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":1.1,"outputPrice":4.4,"cacheReadsPrice":0.275,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"openai/gpt-4.1":{"maxTokens":209516,"contextWindow":1047576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":2,"outputPrice":8,"cacheReadsPrice":0.5,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4.1-mini":{"maxTokens":32768,"contextWindow":1047576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.39999999999999997,"outputPrice":1.5999999999999999,"cacheReadsPrice":0.09999999999999999,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4.1-nano":{"maxTokens":32768,"contextWindow":1047576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"cacheReadsPrice":0.024999999999999998,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-4-maverick":{"maxTokens":16384,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.15,"outputPrice":0.6,"description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-4-scout":{"maxTokens":16384,"contextWindow":10000000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.08,"outputPrice":0.3,"description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"deepseek/deepseek-chat-v3-0324":{"maxTokens":16384,"contextWindow":163840,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.19999999999999998,"outputPrice":0.77,"cacheReadsPrice":0.135,"description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/o1-pro":{"maxTokens":100000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":150,"outputPrice":600,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"mistralai/mistral-small-3.1-24b-instruct":{"maxTokens":128000,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.351,"outputPrice":0.5549999999999999,"description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"google/gemma-3-4b-it":{"maxTokens":16384,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.04,"outputPrice":0.08,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"google/gemma-3-12b-it":{"maxTokens":16384,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.04,"outputPrice":0.13,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"cohere/command-a":{"maxTokens":8192,"contextWindow":256000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":2.5,"outputPrice":10,"description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4o-mini-search-preview":{"maxTokens":16384,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.15,"outputPrice":0.6,"description":"GPT-4o mini Search Preview is a specialized model for web search in Chat Completions. It is trained to understand and execute web search queries.","supportsReasoningEffort":false,"supportedParameters":["max_tokens"]},"openai/gpt-4o-search-preview":{"maxTokens":16384,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":2.5,"outputPrice":10,"description":"GPT-4o Search Previewis a specialized model for web search in Chat Completions. It is trained to understand and execute web search queries.","supportsReasoningEffort":false,"supportedParameters":["max_tokens"]},"rekaai/reka-flash-3":{"maxTokens":65536,"contextWindow":65536,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.09999999999999999,"outputPrice":0.19999999999999998,"description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/gemma-3-27b-it":{"maxTokens":16384,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.08,"outputPrice":0.16,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"thedrummer/skyfall-36b-v2":{"maxTokens":32768,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.55,"outputPrice":0.7999999999999999,"cacheReadsPrice":0.25,"description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"perplexity/sonar-reasoning-pro":{"maxTokens":25600,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":2,"outputPrice":8,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"perplexity/sonar-pro":{"maxTokens":8000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":3,"outputPrice":15,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"perplexity/sonar-deep-research":{"maxTokens":25600,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":2,"outputPrice":8,"description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"google/gemini-2.0-flash-lite-001":{"maxTokens":8192,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.075,"outputPrice":0.3,"description":"Gemini 2.0 Flash Lite offers a significantly faster time to first token (TTFT) compared to [Gemini Flash 1.5](/google/gemini-flash-1.5), while maintaining quality on par with larger models like [Gemini Pro 1.5](/google/gemini-pro-1.5),...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mistralai/mistral-saba":{"maxTokens":6554,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.19999999999999998,"outputPrice":0.6,"cacheReadsPrice":0.02,"description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-guard-3-8b":{"maxTokens":131072,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.48400000000000004,"outputPrice":0.03,"description":"Llama Guard 3 is a Llama-3.1-8B pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM inputs (prompt classification)...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/o3-mini-high":{"maxTokens":100000,"contextWindow":200000,"supportsImages":false,"supportsPromptCache":true,"inputPrice":1.1,"outputPrice":4.4,"cacheReadsPrice":0.55,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"google/gemini-2.0-flash-001":{"maxTokens":8192,"contextWindow":1048576,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"cacheWritesPrice":0.08333333333333334,"cacheReadsPrice":0.024999999999999998,"description":"Gemini Flash 2.0 offers a significantly faster time to first token (TTFT) compared to [Gemini Flash 1.5](/google/gemini-flash-1.5), while maintaining quality on par with larger models like [Gemini Pro 1.5](/google/gemini-pro-1.5). It...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"aion-labs/aion-1.0":{"maxTokens":32768,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":4,"outputPrice":8,"description":"Aion-1.0 is a multi-model system designed for high performance across various tasks, including reasoning and coding. It is built on DeepSeek-R1, augmented with additional models and techniques such as Tree...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"aion-labs/aion-1.0-mini":{"maxTokens":32768,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.7,"outputPrice":1.4,"description":"Aion-1.0-Mini 32B parameter model is a distilled version of the DeepSeek-R1 model, designed for strong performance in reasoning domains such as mathematics, coding, and logic. It is a modified variant...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"aion-labs/aion-rp-llama-3.1-8b":{"maxTokens":32768,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.7999999999999999,"outputPrice":1.5999999999999999,"description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen2.5-vl-72b-instruct":{"maxTokens":26215,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.25,"outputPrice":0.75,"description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen-plus":{"maxTokens":32768,"contextWindow":1000000,"supportsImages":false,"supportsPromptCache":true,"inputPrice":0.26,"outputPrice":0.78,"cacheWritesPrice":0.325,"cacheReadsPrice":0.052000000000000005,"description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/o3-mini":{"maxTokens":100000,"contextWindow":200000,"supportsImages":false,"supportsPromptCache":true,"inputPrice":1.1,"outputPrice":4.4,"cacheReadsPrice":0.55,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"mistralai/mistral-small-24b-instruct-2501":{"maxTokens":16384,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.049999999999999996,"outputPrice":0.08,"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"deepseek/deepseek-r1-distill-qwen-32b":{"maxTokens":32768,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.29,"outputPrice":0.29,"description":"DeepSeek R1 Distill Qwen 32B is a distilled large language model based on [Qwen 2.5 32B](https://huggingface.co/Qwen/Qwen2.5-32B), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). It outperforms OpenAI's o1-mini across various benchmarks, achieving new...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"perplexity/sonar":{"maxTokens":25415,"contextWindow":127072,"supportsImages":true,"supportsPromptCache":false,"inputPrice":1,"outputPrice":1,"description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"deepseek/deepseek-r1-distill-llama-70b":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.7,"outputPrice":0.7999999999999999,"description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"deepseek/deepseek-r1":{"maxTokens":16000,"contextWindow":163840,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.7,"outputPrice":2.5,"description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning","temperature"]},"minimax/minimax-01":{"maxTokens":1000192,"contextWindow":1000192,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.19999999999999998,"outputPrice":1.1,"description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"microsoft/phi-4":{"maxTokens":16384,"contextWindow":16384,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.065,"outputPrice":0.14,"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"sao10k/l3.1-70b-hanami-x1":{"maxTokens":3200,"contextWindow":16000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":3,"outputPrice":3,"description":"This is [Sao10K](/sao10k)'s experiment over [Euryale v2.2](/sao10k/l3.1-euryale-70b).","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"deepseek/deepseek-chat":{"maxTokens":16000,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.2288,"outputPrice":0.9144,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"sao10k/l3.3-euryale-70b":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.65,"outputPrice":0.75,"description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/o1":{"maxTokens":100000,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":15,"outputPrice":60,"cacheReadsPrice":7.5,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","supportsReasoningEffort":true,"supportedParameters":["include_reasoning","max_tokens","reasoning"]},"cohere/command-r7b-12-2024":{"maxTokens":4000,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.0375,"outputPrice":0.15,"description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-3.3-70b-instruct:free":{"maxTokens":26215,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-3.3-70b-instruct":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.09999999999999999,"outputPrice":0.32,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"amazon/nova-lite-v1":{"maxTokens":5120,"contextWindow":300000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.06,"outputPrice":0.24,"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"amazon/nova-micro-v1":{"maxTokens":5120,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.035,"outputPrice":0.14,"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"amazon/nova-pro-v1":{"maxTokens":5120,"contextWindow":300000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.7999999999999999,"outputPrice":3.1999999999999997,"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4o-2024-11-20":{"maxTokens":16384,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":2.5,"outputPrice":10,"cacheReadsPrice":1.25,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mistralai/mistral-large-2407":{"maxTokens":26215,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":true,"inputPrice":2,"outputPrice":6,"cacheReadsPrice":0.19999999999999998,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen-2.5-coder-32b-instruct":{"maxTokens":32768,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.66,"outputPrice":1,"description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"thedrummer/unslopnemo-12b":{"maxTokens":32768,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.39999999999999997,"outputPrice":0.39999999999999997,"description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"anthropic/claude-3.5-haiku":{"maxTokens":8192,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.7999999999999999,"outputPrice":4,"cacheWritesPrice":1,"cacheReadsPrice":0.08,"description":"Claude 3.5 Haiku features offers enhanced capabilities in speed, coding accuracy, and tool use. Engineered to excel in real-time applications, it delivers quick response times that are essential for dynamic...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"anthracite-org/magnum-v4-72b":{"maxTokens":2048,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":false,"inputPrice":3,"outputPrice":5,"description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen-2.5-7b-instruct":{"maxTokens":32768,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.04,"outputPrice":0.09999999999999999,"description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"inflection/inflection-3-productivity":{"maxTokens":1024,"contextWindow":8000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":2.5,"outputPrice":10,"description":"Inflection 3 Productivity is optimized for following instructions. It is better for tasks requiring JSON output or precise adherence to provided guidelines. It has access to recent news. For emotional...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"inflection/inflection-3-pi":{"maxTokens":1024,"contextWindow":8000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":2.5,"outputPrice":10,"description":"Inflection 3 Pi powers Inflection's [Pi](https://pi.ai) chatbot, including backstory, emotional intelligence, productivity, and safety. It has access to recent news, and excels in scenarios like customer support and roleplay. Pi...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"thedrummer/rocinante-12b":{"maxTokens":32768,"contextWindow":32768,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.16999999999999998,"outputPrice":0.43,"description":"Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-3.2-1b-instruct":{"maxTokens":60000,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.027,"outputPrice":0.201,"description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-3.2-3b-instruct:free":{"maxTokens":26215,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-3.2-3b-instruct":{"maxTokens":80000,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.0509,"outputPrice":0.335,"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-3.2-11b-vision-instruct":{"maxTokens":16384,"contextWindow":131072,"supportsImages":true,"supportsPromptCache":false,"inputPrice":0.245,"outputPrice":0.245,"description":"Llama 3.2 11B Vision is a multimodal model with 11 billion parameters, designed to handle tasks combining visual and textual data. It excels in tasks such as image captioning and...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"qwen/qwen-2.5-72b-instruct":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.36,"outputPrice":0.39999999999999997,"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"cohere/command-r-plus-08-2024":{"maxTokens":4000,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":2.5,"outputPrice":10,"description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"cohere/command-r-08-2024":{"maxTokens":4000,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.15,"outputPrice":0.6,"description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"sao10k/l3.1-euryale-70b":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.85,"outputPrice":0.85,"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"nousresearch/hermes-3-llama-3.1-70b":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.3,"outputPrice":0.3,"description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"nousresearch/hermes-3-llama-3.1-405b:free":{"maxTokens":26215,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0,"outputPrice":0,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"nousresearch/hermes-3-llama-3.1-405b":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":1,"outputPrice":1,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"sao10k/l3-lunaris-8b":{"maxTokens":16384,"contextWindow":8192,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.04,"outputPrice":0.049999999999999996,"description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4o-2024-08-06":{"maxTokens":16384,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":2.5,"outputPrice":10,"cacheReadsPrice":1.25,"description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-3.1-70b-instruct":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.39999999999999997,"outputPrice":0.39999999999999997,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-3.1-8b-instruct":{"maxTokens":16384,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.02,"outputPrice":0.049999999999999996,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mistralai/mistral-nemo":{"maxTokens":26215,"contextWindow":131072,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.02,"outputPrice":0.03,"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4o-mini-2024-07-18":{"maxTokens":16384,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.15,"outputPrice":0.6,"cacheReadsPrice":0.075,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4o-mini":{"maxTokens":16384,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.15,"outputPrice":0.6,"cacheReadsPrice":0.075,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"google/gemma-2-27b-it":{"maxTokens":2048,"contextWindow":8192,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.65,"outputPrice":0.65,"description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"sao10k/l3-euryale-70b":{"maxTokens":8192,"contextWindow":8192,"supportsImages":false,"supportsPromptCache":false,"inputPrice":1.48,"outputPrice":1.48,"description":"Euryale 70B v2.1 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). - Better prompt adherence. - Better anatomy / spatial awareness. - Adapts much better to unique and custom...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"nousresearch/hermes-2-pro-llama-3-8b":{"maxTokens":8192,"contextWindow":8192,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.14,"outputPrice":0.14,"description":"Hermes 2 Pro is an upgraded, retrained version of Nous Hermes 2, consisting of an updated and cleaned version of the OpenHermes 2.5 Dataset, as well as a newly introduced...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4o":{"maxTokens":16384,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":2.5,"outputPrice":10,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4o-2024-05-13":{"maxTokens":4096,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":5,"outputPrice":15,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-3-8b-instruct":{"maxTokens":8192,"contextWindow":8192,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.04,"outputPrice":0.04,"description":"Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 8B instruct-tuned version was optimized for high quality dialogue usecases. It has demonstrated strong...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"meta-llama/llama-3-70b-instruct":{"maxTokens":8000,"contextWindow":8192,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.51,"outputPrice":0.74,"description":"Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 70B instruct-tuned version was optimized for high quality dialogue usecases. It has demonstrated strong...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mistralai/mixtral-8x22b-instruct":{"maxTokens":13108,"contextWindow":65536,"supportsImages":false,"supportsPromptCache":true,"inputPrice":2,"outputPrice":6,"cacheReadsPrice":0.19999999999999998,"description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"microsoft/wizardlm-2-8x22b":{"maxTokens":8000,"contextWindow":65536,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.62,"outputPrice":0.62,"description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4-turbo":{"maxTokens":4096,"contextWindow":128000,"supportsImages":true,"supportsPromptCache":false,"inputPrice":10,"outputPrice":30,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"anthropic/claude-3-haiku":{"maxTokens":4096,"contextWindow":200000,"supportsImages":true,"supportsPromptCache":true,"inputPrice":0.25,"outputPrice":1.25,"cacheWritesPrice":0.3,"cacheReadsPrice":0.03,"description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mistralai/mistral-large":{"maxTokens":25600,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":true,"inputPrice":2,"outputPrice":6,"cacheReadsPrice":0.19999999999999998,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-3.5-turbo-0613":{"maxTokens":4096,"contextWindow":4095,"supportsImages":false,"supportsPromptCache":false,"inputPrice":1,"outputPrice":2,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","supportsReasoningEffort":false,"supportedParameters":["temperature"]},"openai/gpt-4-turbo-preview":{"maxTokens":4096,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":10,"outputPrice":30,"description":"The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4-1106-preview":{"maxTokens":4096,"contextWindow":128000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":10,"outputPrice":30,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to April 2023.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-3.5-turbo-instruct":{"maxTokens":4096,"contextWindow":4095,"supportsImages":false,"supportsPromptCache":false,"inputPrice":1.5,"outputPrice":2,"description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-3.5-turbo-16k":{"maxTokens":4096,"contextWindow":16385,"supportsImages":false,"supportsPromptCache":false,"inputPrice":3,"outputPrice":4,"description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"mancer/weaver":{"maxTokens":2000,"contextWindow":8000,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.75,"outputPrice":1,"description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"undi95/remm-slerp-l2-13b":{"maxTokens":4096,"contextWindow":6144,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.44999999999999996,"outputPrice":0.65,"description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"gryphe/mythomax-l2-13b":{"maxTokens":4096,"contextWindow":4096,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.06,"outputPrice":0.06,"description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4-0314":{"maxTokens":4096,"contextWindow":8191,"supportsImages":false,"supportsPromptCache":false,"inputPrice":30,"outputPrice":60,"description":"GPT-4-0314 is the first version of GPT-4 released, with a context length of 8,192 tokens, and was supported until June 14. Training data: up to Sep 2021.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-4":{"maxTokens":4096,"contextWindow":8191,"supportsImages":false,"supportsPromptCache":false,"inputPrice":30,"outputPrice":60,"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]},"openai/gpt-3.5-turbo":{"maxTokens":4096,"contextWindow":16385,"supportsImages":false,"supportsPromptCache":false,"inputPrice":0.5,"outputPrice":1.5,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","supportsReasoningEffort":false,"supportedParameters":["max_tokens","temperature"]}}