{"alibaba/qwen3-max":{"maxTokens":65536,"contextWindow":262144,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.861,"outputPrice":3.4410000000000003,"description":"This is the best-performing model in the Qwen series. It is ideal for complex, multi-step tasks."},"alibaba/qwen3-30b-a3b-instruct-2507":{"maxTokens":65536,"contextWindow":131072,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":0.7999999999999999,"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and agentic tool use. Post-trained on instruction data, it demonstrates competitive performance across reasoning (AIME, ZebraLogic), coding (MultiPL-E, LiveCodeBench), and alignment (IFEval, WritingBench) benchmarks. It outperforms its non-instruct variant on subjective and open-ended tasks while retaining strong factual and coding performance.","cacheWritesPrice":0.7999999999999999,"cacheReadsPrice":0.19999999999999998},"alibaba/qwen-plus":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.2,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.39999999999999997,"cacheReadsPrice":0.39999999999999997},"alibaba/qwen3-coder-plus":{"maxTokens":65536,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":""},"alibaba/qwen-turbo":{"maxTokens":0,"contextWindow":1000000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.049999999999999996,"outputPrice":0.19999999999999998,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.049999999999999996,"cacheReadsPrice":0.049999999999999996},"alibaba/qwen-max":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.5999999999999999,"outputPrice":6.3999999999999995,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":1.5999999999999999,"cacheReadsPrice":1.5999999999999999},"alibaba/qwen3-coder-flash":{"maxTokens":65536,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":1.5,"description":"","cacheWritesPrice":0.3,"cacheReadsPrice":0.08},"google/gemini-1.5-flash-8b-latest":{"maxTokens":8192,"contextWindow":1048576,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.0375,"outputPrice":0.15,"description":"","cacheWritesPrice":0.0375,"cacheReadsPrice":0.0375},"google/gemini-1.5-pro":{"maxTokens":8192,"contextWindow":2097152,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":5,"description":"","cacheWritesPrice":1.25,"cacheReadsPrice":1.25},"google/gemini-1.5-flash":{"maxTokens":8192,"contextWindow":1048576,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.075,"outputPrice":0.3,"description":"","cacheWritesPrice":0.075,"cacheReadsPrice":0.075},"google/gemini-1.5-flash-8b":{"maxTokens":8192,"contextWindow":1048576,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.075,"outputPrice":0.3,"description":"","cacheWritesPrice":0.075,"cacheReadsPrice":0.075},"google/gemini-2.5-flash":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"google/gemini-1.5-flash-latest":{"maxTokens":8192,"contextWindow":1048576,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.075,"outputPrice":0.3,"description":"","cacheWritesPrice":0.075,"cacheReadsPrice":0.075},"google/gemini-1.5-pro-latest":{"maxTokens":8192,"contextWindow":2097152,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":5,"description":"","cacheWritesPrice":1.25,"cacheReadsPrice":1.25},"google/gemini-2.5-pro":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"google/gemini-2.5-flash-lite-preview-06-17":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"description":"Google's smallest and most cost effective model, built for at scale usage.","cacheWritesPrice":0.35,"cacheReadsPrice":0.024999999999999998},"google/gemini-2.0-flash-001":{"maxTokens":8192,"contextWindow":1048576,"supportsPromptCache":false,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"description":"Gemini Flash 2.0 offers a significantly faster time to first token (TTFT) compared to [Gemini Flash 1.5](/google/gemini-flash-1.5), while maintaining quality on par with larger models like [Gemini Pro 1.5](/google/gemini-pro-1.5). It introduces notable enhancements in multimodal understanding, coding capabilities, complex instruction following, and function calling. These advancements come together to deliver more seamless and robust agentic experiences.","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.09999999999999999},"parasail/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"maxTokens":1048576,"contextWindow":1048576,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.21,"outputPrice":0.85,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.85,"cacheReadsPrice":0.21},"parasail/parasail-qwen3-235b-a22b-instruct-2507":{"maxTokens":8192,"contextWindow":262144,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.15,"outputPrice":0.85,"description":"","cacheWritesPrice":0.85,"cacheReadsPrice":0.15},"parasail/parasail-mistral-7b-instruct-03":{"maxTokens":8192,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.11,"outputPrice":0.11,"description":"","cacheWritesPrice":0.11,"cacheReadsPrice":0.11},"parasail/parasail-glm-45":{"maxTokens":8192,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.59,"outputPrice":2.0999999999999996,"description":"","cacheWritesPrice":2.0999999999999996,"cacheReadsPrice":0.59},"parasail/parasail-deepseek-r1":{"maxTokens":8192,"contextWindow":64000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":3,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":3,"cacheReadsPrice":3},"parasail/parasail-qwen3-coder-480b-a35b-instruct":{"maxTokens":8192,"contextWindow":264144,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":3.5,"description":"","cacheWritesPrice":3.5,"cacheReadsPrice":2},"parasail/parasail-deepseek-v3-0324":{"maxTokens":0,"contextWindow":163840,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7899999999999999,"outputPrice":1.15,"description":"","cacheWritesPrice":1.15,"cacheReadsPrice":0.7899999999999999},"parasail/parasail-kimi-k2-instruct":{"maxTokens":16384,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.9900000000000001,"outputPrice":2.99,"description":"","cacheWritesPrice":2.99,"cacheReadsPrice":0.9900000000000001},"parasail/parasail-eva-25-72b-v02-fp8":{"maxTokens":8192,"contextWindow":32000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7,"outputPrice":0.7,"description":"","cacheWritesPrice":0.7,"cacheReadsPrice":0.7},"parasail/parasail-qwen25-vl-72b-instruct":{"maxTokens":8192,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7,"outputPrice":0.7,"description":"","cacheWritesPrice":0.7,"cacheReadsPrice":0.7},"parasail/parasail-mistral-nemo":{"maxTokens":8192,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.11,"outputPrice":0.11,"description":"","cacheWritesPrice":0.11,"cacheReadsPrice":0.11},"parasail/meta-llama/Llama-4-Scout-17B-16E-Instruct":{"maxTokens":1048576,"contextWindow":1048576,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.14,"outputPrice":0.58,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.58,"cacheReadsPrice":0.14},"parasail/parasail-anubis-pro":{"maxTokens":8192,"contextWindow":64000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7999999999999999,"outputPrice":0.7999999999999999,"description":"","cacheWritesPrice":0.7999999999999999,"cacheReadsPrice":0.7999999999999999},"parasail/parasail-gemma3-27b-it":{"maxTokens":8192,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":0.5,"description":"Gemma 3 1B is the smallest of the new Gemma 3 family. It handles context windows up to 32k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities, including structured outputs and function calling. Note: Gemma 3 1B is not multimodal. For the smallest multimodal Gemma 3 model, please see [Gemma 3 4B](google/gemma-3-4b-it)","cacheWritesPrice":0.5,"cacheReadsPrice":0.3},"perplexity/sonar":{"maxTokens":8192,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":1,"description":"Lightweight offering with search grounding, quicker and cheaper than Sonar Pro.","cacheWritesPrice":1,"cacheReadsPrice":1},"perplexity/sonar-pro":{"maxTokens":8192,"contextWindow":204800,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Premier search offering with search grounding, supporting advanced queries and follow-ups.","cacheWritesPrice":3,"cacheReadsPrice":3},"perplexity/sonar-reasoning-pro":{"maxTokens":8192,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":8,"description":"Premier reasoning offering powered by DeepSeek R1 with Chain of Thought (CoT).","cacheWritesPrice":2,"cacheReadsPrice":2},"minimaxi/MiniMax-M2":{"maxTokens":128000,"contextWindow":200000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning, tool use, and multi-step task execution while maintaining low latency and deployment efficiency."},"bedrock/claude-sonnet-4@eu-north-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-3-7-sonnet@us-west-2":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-sonnet-4-5@eu-west-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-3-7-sonnet@eu-west-3":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-opus-4@us-west-2":{"maxTokens":32000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Claude Opus 4 is Anthropic's most powerful model yet and the best coding model in the world, leading on SWE-bench (72.5%) and Terminal-bench (43.2%). It delivers sustained performance on long-running tasks that require focused effort and thousands of steps, with the ability to work continuously for several hours—dramatically outperforming all Sonnet models and significantly expanding what AI agents can accomplish.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"bedrock/claude-sonnet-4@us-east-2":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-sonnet-4":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-opus-4@us-east-2":{"maxTokens":32000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Claude Opus 4 is Anthropic's most powerful model yet and the best coding model in the world, leading on SWE-bench (72.5%) and Terminal-bench (43.2%). It delivers sustained performance on long-running tasks that require focused effort and thousands of steps, with the ability to work continuously for several hours—dramatically outperforming all Sonnet models and significantly expanding what AI agents can accomplish.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"bedrock/claude-sonnet-4@us-east-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-haiku-4-5@us-west-2":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"bedrock/claude-3-7-sonnet@eu-west-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-haiku-4-5@us-east-2":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"bedrock/claude-3-7-sonnet@us-east-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-3-7-sonnet@eu-north-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-sonnet-4-5@us-west-2":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-haiku-4-5@us-east-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"bedrock/claude-3-7-sonnet@us-east-2":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-haiku-4-5@eu-central-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"bedrock/claude-sonnet-4@eu-central-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-opus-4":{"maxTokens":32000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Claude Opus 4 is Anthropic's most powerful model yet and the best coding model in the world, leading on SWE-bench (72.5%) and Terminal-bench (43.2%). It delivers sustained performance on long-running tasks that require focused effort and thousands of steps, with the ability to work continuously for several hours—dramatically outperforming all Sonnet models and significantly expanding what AI agents can accomplish.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"bedrock/claude-sonnet-4-5@eu-central-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-sonnet-4-5@us-east-2":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-haiku-4-5":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"bedrock/claude-haiku-4-5@eu-west-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"bedrock/claude-sonnet-4-5@eu-north-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-haiku-4-5@eu-north-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"bedrock/claude-sonnet-4-5@us-east-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-3-7-sonnet@eu-central-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-sonnet-4-5":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-sonnet-4-5@eu-west-3":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-sonnet-4@eu-west-3":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-haiku-4-5@eu-west-3":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"bedrock/claude-3-7-sonnet":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-sonnet-4@us-west-2":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-sonnet-4@eu-west-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"bedrock/claude-opus-4@us-east-1":{"maxTokens":32000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Claude Opus 4 is Anthropic's most powerful model yet and the best coding model in the world, leading on SWE-bench (72.5%) and Terminal-bench (43.2%). It delivers sustained performance on long-running tasks that require focused effort and thousands of steps, with the ability to work continuously for several hours—dramatically outperforming all Sonnet models and significantly expanding what AI agents can accomplish.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"mistral/pixtral-large-latest":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":5,"description":"","cacheWritesPrice":5,"cacheReadsPrice":2},"mistral/mistral-large-latest":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":6,"description":"","cacheWritesPrice":2,"cacheReadsPrice":2},"mistral/devstral-small-2507":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.3,"description":"","cacheWritesPrice":0.3,"cacheReadsPrice":0.09999999999999999},"mistral/devstral-medium-2507":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":2,"description":"","cacheWritesPrice":2,"cacheReadsPrice":0.39999999999999997},"mistral/codestral-latest":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":0.8999999999999999,"description":"","cacheWritesPrice":0.8999999999999999,"cacheReadsPrice":0.3},"mistral/mistral-small-latest":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.3,"description":"","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.09999999999999999},"mistral/open-mistral-7b":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.25,"outputPrice":0.25,"description":"","cacheWritesPrice":0.25,"cacheReadsPrice":0.25},"mistral/mistral-medium-latest":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":2,"description":"","cacheWritesPrice":2,"cacheReadsPrice":0.39999999999999997},"mistral/magistral-medium-latest":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":5,"description":"","cacheWritesPrice":5,"cacheReadsPrice":2},"mistral/devstral-small-latest":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.3,"description":"","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.09999999999999999},"deepinfra/Qwen/Qwen2.5-Coder-32B-Instruct":{"maxTokens":0,"contextWindow":16384,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.07,"outputPrice":0.16,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.07,"cacheReadsPrice":0.07},"deepinfra/microsoft/phi-4":{"maxTokens":0,"contextWindow":16384,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.07,"outputPrice":0.14,"description":"Phi-4-reasoning-plus is an enhanced 14B parameter model from Microsoft, fine-tuned from Phi-4 with additional reinforcement learning to boost accuracy on math, science, and code reasoning tasks. It uses the same dense decoder-only transformer architecture as Phi-4, but generates longer, more comprehensive outputs structured into a step-by-step reasoning trace and final answer.\n\nWhile it offers improved benchmark scores over Phi-4-reasoning across tasks like AIME, OmniMath, and HumanEvalPlus, its responses are typically ~50% longer, resulting in higher latency. Designed for English-only applications, it is well-suited for structured reasoning workflows where output quality takes priority over response speed.","cacheWritesPrice":0.07,"cacheReadsPrice":0.07},"deepinfra/meta-llama/Llama-3.3-70B-Instruct":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.22999999999999998,"outputPrice":0.39999999999999997,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.22999999999999998,"cacheReadsPrice":0.22999999999999998},"deepinfra/Qwen/Qwen3-235B-A22B":{"maxTokens":4096,"contextWindow":40000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":0.6,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.19999999999999998,"cacheReadsPrice":0.19999999999999998},"deepinfra/Qwen/Qwen3-32B":{"maxTokens":0,"contextWindow":40000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.3,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.09999999999999999},"deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct":{"maxTokens":0,"contextWindow":262144,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.5999999999999999,"description":"","cacheWritesPrice":1.5999999999999999,"cacheReadsPrice":0.39999999999999997},"deepinfra/Qwen/Qwen2.5-72B-Instruct":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.22999999999999998,"outputPrice":0.39999999999999997,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.22999999999999998,"cacheReadsPrice":0.22999999999999998},"deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.12,"outputPrice":0.3,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.12,"cacheReadsPrice":0.12},"deepinfra/Qwen/QwQ-32B":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.12,"outputPrice":0.18,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.12,"cacheReadsPrice":0.12},"deepinfra/zai-org/GLM-4.5-Air":{"maxTokens":4096,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":1.1,"description":"The GLM-4.5 series models are foundation models designed for intelligent agents. GLM-4.5 has 355 billion total parameters with 32 billion active parameters, while GLM-4.5-Air adopts a more compact design with 106 billion total parameters and 12 billion active parameters. GLM-4.5 models unify reasoning, coding, and intelligent agent capabilities to meet the complex demands of intelligent agent applications.","cacheWritesPrice":1.1,"cacheReadsPrice":0.19999999999999998},"deepinfra/deepseek-ai/DeepSeek-R1":{"maxTokens":8192,"contextWindow":64000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.85,"outputPrice":2.5,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":0.85,"cacheReadsPrice":0.85},"deepinfra/meta-llama/Meta-Llama-3.1-405B-Instruct":{"maxTokens":0,"contextWindow":130815,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7999999999999999,"outputPrice":0.7999999999999999,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.7999999999999999,"cacheReadsPrice":0.7999999999999999},"deepinfra/deepseek-ai/DeepSeek-V3.1":{"maxTokens":0,"contextWindow":163840,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":1,"description":"","cacheWritesPrice":1,"cacheReadsPrice":0.3},"deepinfra/meta-llama/Llama-3.2-90B-Vision-Instruct":{"maxTokens":4096,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.35,"outputPrice":0.39999999999999997,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.35,"cacheReadsPrice":0.35},"deepinfra/deepseek-ai/DeepSeek-V3":{"maxTokens":8192,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.85,"outputPrice":0.8999999999999999,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":0.85,"cacheReadsPrice":0.85},"deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.02,"outputPrice":0.049999999999999996,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.02,"cacheReadsPrice":0.02},"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B":{"maxTokens":8192,"contextWindow":64000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.22999999999999998,"outputPrice":0.69,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":0.22999999999999998,"cacheReadsPrice":0.22999999999999998},"deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct":{"maxTokens":0,"contextWindow":130815,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.22999999999999998,"outputPrice":0.39999999999999997,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.22999999999999998,"cacheReadsPrice":0.22999999999999998},"deepinfra/zai-org/GLM-4.5":{"maxTokens":4096,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.6,"outputPrice":2.2,"description":"The GLM-4.5 series models are foundation models designed for intelligent agents. GLM-4.5 has 355 billion total parameters with 32 billion active parameters, while GLM-4.5-Air adopts a more compact design with 106 billion total parameters and 12 billion active parameters. GLM-4.5 models unify reasoning, coding, and intelligent agent capabilities to meet the complex demands of intelligent agent applications.","cacheWritesPrice":2.2,"cacheReadsPrice":0.6},"nebius/meta-llama/Llama-3.3-70B-Instruct":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.13,"outputPrice":0.39999999999999997,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.13,"cacheReadsPrice":0.13},"nebius/Qwen/QwQ-32B":{"maxTokens":0,"contextWindow":32000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.15,"outputPrice":0.44999999999999996,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.15,"cacheReadsPrice":0.15},"nebius/deepseek-ai/DeepSeek-R1":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7999999999999999,"outputPrice":2.4,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":0.7999999999999999,"cacheReadsPrice":0.7999999999999999},"nebius/moonshotai/Kimi-K2-Instruct":{"maxTokens":0,"contextWindow":131000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.5,"outputPrice":2.4,"description":"","cacheWritesPrice":2.4,"cacheReadsPrice":0.5},"nebius/Qwen/Qwen3-Coder-480B-A35B-Instruct":{"maxTokens":0,"contextWindow":262000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.7999999999999998,"description":"","cacheWritesPrice":1.7999999999999998,"cacheReadsPrice":0.39999999999999997},"nebius/deepseek-ai/DeepSeek-R1-0528":{"maxTokens":0,"contextWindow":164000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7999999999999999,"outputPrice":2.4,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":0.7999999999999999,"cacheReadsPrice":0.7999999999999999},"nebius/Qwen/QwQ-32B-fast":{"maxTokens":0,"contextWindow":32000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.5,"outputPrice":1.5,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.5,"cacheReadsPrice":0.5},"nebius/deepseek-ai/DeepSeek-V3":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.5,"outputPrice":1.5,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":0.5,"cacheReadsPrice":0.5},"nebius/deepseek-ai/DeepSeek-R1-fast":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":6,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":2,"cacheReadsPrice":2},"nebius/zai-org/GLM-4.5":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.6,"outputPrice":2.2,"description":"","cacheWritesPrice":2.2,"cacheReadsPrice":0.6},"nebius/deepseek-ai/DeepSeek-V3-0324":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.5,"outputPrice":1.5,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":0.5,"cacheReadsPrice":0.5},"nebius/deepseek-ai/DeepSeek-V3-0324-fast":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":6,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":6,"cacheReadsPrice":2},"groq/qwen-qwq-32b":{"maxTokens":131072,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.29,"outputPrice":0.39,"description":"Qwen/QwQ-32B is a breakthrough 32-billion parameter reasoning model delivering performance comparable to state-of-the-art (SOTA) models 20x larger like DeepSeek-R1 (671B parameters) on complex reasoning and coding tasks. Deployed on Groq's hardware, it provides the world's fastest and cost-efficient reasoning, producing chains and results in seconds. Along with native tool use support, the 128K context window enables processing extensive information while maintaining comprehensive context.","cacheWritesPrice":0.29,"cacheReadsPrice":0.29},"groq/moonshotai/Kimi-K2-Instruct-0905":{"maxTokens":16384,"contextWindow":256000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":3,"description":"Moonshot AI’s cutting‑edge model, moonshotai/Kimi-K2-Instruct-0905, is now live on GroqCloud.","cacheWritesPrice":3,"cacheReadsPrice":1},"groq/moonshotai/kimi-k2-instruct":{"maxTokens":16384,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":3,"description":"","cacheWritesPrice":3,"cacheReadsPrice":1},"groq/openai/gpt-oss-20b":{"maxTokens":32768,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.5,"description":"","cacheWritesPrice":0.5,"cacheReadsPrice":0.09999999999999999},"groq/openai/gpt-oss-120b":{"maxTokens":32768,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.15,"outputPrice":0.75,"description":"","cacheWritesPrice":0.75,"cacheReadsPrice":0.15},"vertex/claude-haiku-4-5@us-east5":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"vertex/claude-sonnet-4-5@europe-west1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"vertex/google/gemini-2.5-flash@us-east1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"vertex/google/gemini-2.5-pro@europe-north1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"vertex/google/gemini-2.5-pro@europe-west8":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"vertex/google/gemini-2.5-flash@us-east5":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"vertex/google/gemini-2.5-pro@europe-west4":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"vertex/google/gemini-2.5-flash-image-preview":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":30,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":30,"cacheReadsPrice":0.3},"vertex/google/gemini-2.5-flash@europe-west1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"vertex/claude-sonnet-4":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"vertex/claude-sonnet-4@europe-west1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"vertex/google/gemini-2.5-flash@europe-north1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"vertex/google/gemini-2.5-flash@europe-west8":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"vertex/google/gemini-2.5-pro@us-east5":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"vertex/google/gemini-2.5-pro@europe-west1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"vertex/claude-sonnet-4-5":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"vertex/claude-opus-4":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"vertex/claude-opus-4-1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"vertex/google/gemini-2.5-flash@europe-central2":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"vertex/google/gemini-2.5-pro@us-east1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"vertex/claude-3-5-sonnet@us-east5":{"maxTokens":8192,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's previous most intelligent model. High level of intelligence and capability. Excells in coding.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"vertex/claude-3-7-sonnet@us-east5":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"vertex/claude-opus-4-1@us-east5":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"vertex/google/gemini-2.5-pro@us-west1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"vertex/claude-haiku-4-5@europe-west1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"vertex/claude-sonnet-4@us-east5":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"vertex/google/gemini-2.5-pro@europe-central2":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"vertex/claude-3-7-sonnet@europe-west1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"vertex/claude-sonnet-4-5@us-east5":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"vertex/claude-opus-4@us-east5":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"vertex/google/gemini-2.5-flash@us-central1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"vertex/google/gemini-2.5-pro@us-central1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"vertex/claude-haiku-4-5":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"vertex/claude-opus-4@europe-west1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"vertex/claude-opus-4-1@europe-west1":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"vertex/google/gemini-2.5-flash@us-south1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"vertex/google/gemini-2.5-flash@us-west1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"vertex/google/gemini-2.5-flash@europe-west4":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"vertex/claude-3-5-sonnet":{"maxTokens":8192,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's previous most intelligent model. High level of intelligence and capability. Excells in coding.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"vertex/google/gemini-2.5-pro":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"vertex/google/gemini-2.5-pro@us-south1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"vertex/claude-3-5-sonnet@europe-west1":{"maxTokens":8192,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's previous most intelligent model. High level of intelligence and capability. Excells in coding.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"vertex/google/gemini-2.5-flash":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"vertex/claude-3-7-sonnet":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"ionstream/moonshotai/Kimi-K2-Instruct-0905":{"maxTokens":16384,"contextWindow":256000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":1.25,"description":"","cacheWritesPrice":0.3,"cacheReadsPrice":0.3},"azure/o4-mini@francecentral":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.275},"azure/gpt-4.1@francecentral":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":8,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","cacheWritesPrice":2,"cacheReadsPrice":0.5},"azure/gpt-4.1-nano@westus3":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.024999999999999998},"azure/gpt-5@swedencentral":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"","cacheWritesPrice":10,"cacheReadsPrice":0.125},"azure/gpt-5-mini@eastus2":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.25,"outputPrice":2,"description":"","cacheWritesPrice":0.25,"cacheReadsPrice":0.024999999999999998},"azure/gpt-5-mini@uksouth":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.25,"outputPrice":2,"description":"","cacheWritesPrice":0.25,"cacheReadsPrice":0.024999999999999998},"azure/gpt-5@uksouth":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"","cacheWritesPrice":10,"cacheReadsPrice":0.125},"azure/gpt-4.1-mini@eastus2":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.5999999999999999,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","cacheWritesPrice":0.39999999999999997,"cacheReadsPrice":0.09999999999999999},"azure/gpt-4.1-mini@swedencentral":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.5999999999999999,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","cacheWritesPrice":0.39999999999999997,"cacheReadsPrice":0.09999999999999999},"azure/gpt-4.1@westus3":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":8,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","cacheWritesPrice":2,"cacheReadsPrice":0.5},"azure/gpt-4.1@swedencentral":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":8,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","cacheWritesPrice":2,"cacheReadsPrice":0.5},"azure/o4-mini":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.275},"azure/gpt-4.1-nano@uksouth":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.024999999999999998},"azure/gpt-4.1-mini@westus3":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.5999999999999999,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","cacheWritesPrice":0.39999999999999997,"cacheReadsPrice":0.09999999999999999},"azure/gpt-4.1-mini@francecentral":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.5999999999999999,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","cacheWritesPrice":0.39999999999999997,"cacheReadsPrice":0.09999999999999999},"azure/gpt-4.1@uksouth":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":8,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","cacheWritesPrice":2,"cacheReadsPrice":0.5},"azure/gpt-4.1-nano":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.024999999999999998},"azure/gpt-4.1-nano@francecentral":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.024999999999999998},"azure/o4-mini@swedencentral":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.275},"azure/gpt-5-nano@swedencentral":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.049999999999999996,"outputPrice":0.39999999999999997,"description":"","cacheWritesPrice":0.049999999999999996,"cacheReadsPrice":0.005},"azure/gpt-5-nano@uksouth":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.049999999999999996,"outputPrice":0.39999999999999997,"description":"","cacheWritesPrice":0.049999999999999996,"cacheReadsPrice":0.005},"azure/gpt-4.1-nano@eastus2":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.024999999999999998},"azure/o4-mini@eastus2":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.275},"azure/gpt-4.1-mini":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.5999999999999999,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","cacheWritesPrice":0.39999999999999997,"cacheReadsPrice":0.09999999999999999},"azure/gpt-4.1":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":8,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","cacheWritesPrice":2,"cacheReadsPrice":0.5},"azure/gpt-4.1@eastus2":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":8,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","cacheWritesPrice":2,"cacheReadsPrice":0.5},"azure/gpt-5-mini":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.25,"outputPrice":2,"description":"","cacheWritesPrice":0.25,"cacheReadsPrice":0.024999999999999998},"azure/gpt-5-mini@swedencentral":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.25,"outputPrice":2,"description":"","cacheWritesPrice":0.25,"cacheReadsPrice":0.024999999999999998},"azure/gpt-5@eastus2":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"","cacheWritesPrice":10,"cacheReadsPrice":0.125},"azure/o4-mini@westus3":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.275},"azure/o4-mini@uksouth":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.275},"azure/gpt-5-nano":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.049999999999999996,"outputPrice":0.39999999999999997,"description":"","cacheWritesPrice":0.049999999999999996,"cacheReadsPrice":0.005},"azure/gpt-4.1-mini@uksouth":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.5999999999999999,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","cacheWritesPrice":0.39999999999999997,"cacheReadsPrice":0.09999999999999999},"azure/gpt-5-nano@eastus2":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.049999999999999996,"outputPrice":0.39999999999999997,"description":"","cacheWritesPrice":0.049999999999999996,"cacheReadsPrice":0.005},"azure/gpt-4.1-nano@swedencentral":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.024999999999999998},"azure/gpt-5":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"","cacheWritesPrice":10,"cacheReadsPrice":0.125},"novita/sophosympatheia/midnight-rose-70b":{"maxTokens":0,"contextWindow":4096,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7999999999999999,"outputPrice":0.7999999999999999,"description":"A merge with a complex family tree, this model was crafted for roleplaying and storytelling. Midnight Rose is a successor to Rogue Rose and Aurora Nights and improves upon them both. It wants to produce lengthy output by default and is the best creative writing merge produced so far by sophosympatheia.","cacheWritesPrice":0.7999999999999999,"cacheReadsPrice":0.7999999999999999},"novita/meta-llama/llama-3-70b-instruct":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.51,"outputPrice":0.74,"description":"Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 70B instruct-tuned version was optimized for high quality dialogue usecases. It has demonstrated strong performance compared to leading closed-source models in human evaluations.","cacheWritesPrice":0.51,"cacheReadsPrice":0.51},"novita/deepseek/deepseek-r1-distill-qwen-14b":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.15,"outputPrice":0.15,"description":"DeepSeek R1 Distill Qwen 14B is a distilled large language model based on Qwen 2.5 14B, using outputs from DeepSeek R1. It outperforms OpenAI's o1-mini across various benchmarks, achieving new state-of-the-art results for dense models.\n\nOther benchmark results include:\n\nAIME 2024 pass@1: 69.7\nMATH-500 pass@1: 93.9\nCodeForces Rating: 1481\nThe model leverages fine-tuning from DeepSeek R1's outputs, enabling competitive performance comparable to larger frontier models.","cacheWritesPrice":0.15,"cacheReadsPrice":0.15},"novita/zai-org/glm-4.6":{"maxTokens":131072,"contextWindow":204800,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.6,"outputPrice":2.2,"description":"GLM-4.6 is Z AI’s latest flagship model, designed to push agentic and coding performance further. It expands the context window from 128K to 200K tokens, improves reasoning and tool-use capabilities, and delivers stronger results in coding benchmarks and real-world development workflows. GLM-4.6 demonstrates refined writing quality, more capable agent behavior, and higher token efficiency (≈15% fewer tokens vs. GLM-4.5).\n\nEvaluations show clear gains over GLM-4.5 across reasoning, agents, and coding, reaching near parity with Claude Sonnet 4 in practical tasks while outperforming other open-source baselines. GLM-4.6 is available through the Z.ai API platform, OpenRouter, coding agents (Claude Code, Roo Code, Cline, Kilo Code), and soon as downloadable weights on HuggingFace and ModelScope.","cacheWritesPrice":2.2,"cacheReadsPrice":0.6},"novita/qwen/qwen-2.5-72b-instruct":{"maxTokens":0,"contextWindow":32000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.38,"outputPrice":0.39999999999999997,"description":"Qwen2.5 is the latest series of Qwen large language models. For Qwen2.5, we release a number of base language models and instruction-tuned language models ranging from 0.5 to 72 billion parameters.","cacheWritesPrice":0.38,"cacheReadsPrice":0.38},"novita/meta-llama/llama-3.2-1b-instruct":{"maxTokens":0,"contextWindow":131000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.02,"outputPrice":0.02,"description":"The Meta Llama 3.2 collection of multilingual large language models (LLMs) is a collection of pretrained and instruction-tuned generative models in 1B and 3B sizes (text in/text out).","cacheWritesPrice":0.02,"cacheReadsPrice":0.02},"novita/deepseek/deepseek_v3":{"maxTokens":0,"contextWindow":64000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.8899999999999999,"outputPrice":0.8899999999999999,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations reveal that the model outperforms other open-source models and rivals leading closed-source models.","cacheWritesPrice":0.8899999999999999,"cacheReadsPrice":0.8899999999999999},"novita/microsoft/wizardlm-2-8x22b":{"maxTokens":0,"contextWindow":65535,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.62,"outputPrice":0.62,"description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models.","cacheWritesPrice":0.62,"cacheReadsPrice":0.62},"novita/meta-llama/llama-3.1-8b-instruct":{"maxTokens":0,"contextWindow":16384,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.049999999999999996,"outputPrice":0.049999999999999996,"description":"Meta's latest class of models, Llama 3.1, launched with a variety of sizes and configurations. The 8B instruct-tuned version is particularly fast and efficient. It has demonstrated strong performance in human evaluations, outperforming several leading closed-source models.","cacheWritesPrice":0.049999999999999996,"cacheReadsPrice":0.049999999999999996},"novita/meta-llama/llama-3-8b-instruct":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.04,"outputPrice":0.04,"description":"Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 8B instruct-tuned version was optimized for high quality dialogue usecases. It has demonstrated strong performance compared to leading closed-source models in human evaluations.","cacheWritesPrice":0.04,"cacheReadsPrice":0.04},"novita/gryphe/mythomax-l2-13b":{"maxTokens":0,"contextWindow":4096,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09,"outputPrice":0.09,"description":"The idea behind this merge is that each layer is composed of several tensors, which are in turn responsible for specific functions. Using MythoLogic-L2's robust understanding as its input and Huginn's extensive writing capability as its output seems to have resulted in a model that exceeds at both, confirming my theory. (More details to be released at a later time).","cacheWritesPrice":0.09,"cacheReadsPrice":0.09},"novita/nousresearch/nous-hermes-llama2-13b":{"maxTokens":0,"contextWindow":4096,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.16999999999999998,"outputPrice":0.16999999999999998,"description":"Nous-Hermes-Llama2-13b is a state-of-the-art language model fine-tuned on over 300,000 instructions. This model was fine-tuned by Nous Research, with Teknium and Emozilla leading the fine tuning process and dataset curation, Redmond AI sponsoring the compute, and several other contributors.","cacheWritesPrice":0.16999999999999998,"cacheReadsPrice":0.16999999999999998},"novita/meta-llama/llama-3.2-11b-vision-instruct":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.06,"outputPrice":0.06,"description":"Llama 3.2 11B Vision is a multimodal model with 11 billion parameters, designed to handle tasks combining visual and textual data. It excels in tasks such as image captioning and visual question answering, bridging the gap between language generation and visual reasoning. Pre-trained on a massive dataset of image-text pairs, it performs well in complex, high-accuracy image analysis. Its ability to integrate visual understanding with language processing makes it an ideal solution for industries requiring comprehensive visual-linguistic AI applications, such as content creation, AI-driven customer service, and research.","cacheWritesPrice":0.06,"cacheReadsPrice":0.06},"novita/zai-org/glm-4.5":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.6,"outputPrice":2.2,"description":"","cacheWritesPrice":2.2,"cacheReadsPrice":0.6},"novita/meta-llama/llama-4-maverick-17b-128e-instruct-fp8":{"maxTokens":1048576,"contextWindow":1048576,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":0.85,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.19999999999999998,"cacheReadsPrice":0.19999999999999998},"novita/deepseek/deepseek-v3-0324":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.3,"description":"DeepSeek R1 is the latest open-source model released by the DeepSeek team, featuring impressive reasoning capabilities, particularly achieving performance comparable to OpenAI's o1 model in mathematics, coding, and reasoning tasks.","cacheWritesPrice":0.39999999999999997,"cacheReadsPrice":0.39999999999999997},"novita/google/gemma-2-9b-it":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.08,"outputPrice":0.08,"description":"Gemma 2 9B by Google is an advanced, open-source language model that sets a new standard for efficiency and performance in its size class.\nDesigned for a wide variety of tasks, it empowers developers and researchers to build innovative applications, while maintaining accessibility, safety, and cost-effectiveness.","cacheWritesPrice":0.08,"cacheReadsPrice":0.08},"novita/openchat/openchat-7b":{"maxTokens":0,"contextWindow":4096,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.06,"outputPrice":0.06,"description":"OpenChat 7B is a library of open-source language models, fine-tuned with \"C-RLFT (Conditioned Reinforcement Learning Fine-Tuning)\" - a strategy inspired by offline reinforcement learning. It has been trained on mixed-quality data without preference labels.","cacheWritesPrice":0.06,"cacheReadsPrice":0.06},"novita/deepseek/deepseek-v3-turbo":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.3,"description":"DeepSeek R1 is the latest open-source model released by the DeepSeek team, featuring impressive reasoning capabilities, particularly achieving performance comparable to OpenAI's o1 model in mathematics, coding, and reasoning tasks.","cacheWritesPrice":0.39999999999999997,"cacheReadsPrice":0.39999999999999997},"novita/deepseek/deepseek-r1":{"maxTokens":0,"contextWindow":64000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":4,"outputPrice":4,"description":"DeepSeek R1 is the latest open-source model released by the DeepSeek team, featuring impressive reasoning capabilities, particularly achieving performance comparable to OpenAI's o1 model in mathematics, coding, and reasoning tasks.","cacheWritesPrice":4,"cacheReadsPrice":4},"novita/qwen/qwen3-235b-a22b-fp8":{"maxTokens":0,"contextWindow":128000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":0.7999999999999999,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.19999999999999998,"cacheReadsPrice":0.19999999999999998},"novita/qwen/qwen-2-7b-instruct":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.054,"outputPrice":0.054,"description":"Qwen2 is the newest series in the Qwen large language model family. Qwen2 7B is a transformer-based model that demonstrates exceptional performance in language understanding, multilingual capabilities, programming, mathematics, and reasoning.","cacheWritesPrice":0.054,"cacheReadsPrice":0.054},"novita/nousresearch/hermes-2-pro-llama-3-8b":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.14,"outputPrice":0.14,"description":"Hermes 2 Pro is an upgraded, retrained version of Nous Hermes 2, consisting of an updated and cleaned version of the OpenHermes 2.5 Dataset, as well as a newly introduced Function Calling and JSON Mode dataset developed in-house.","cacheWritesPrice":0.14,"cacheReadsPrice":0.14},"novita/jondurbin/airoboros-l2-70b":{"maxTokens":0,"contextWindow":4096,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.5,"outputPrice":0.5,"description":"This is a fine-tuned Llama-2 model designed to support longer and more detailed writing prompts, as well as next-chapter generation. It also includes an experimental role-playing instruction set with multi-round dialogues, character interactions, and varying numbers of participants","cacheWritesPrice":0.5,"cacheReadsPrice":0.5},"novita/mistralai/mistral-7b-instruct":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.059,"outputPrice":0.059,"description":"A high-performing, industry-standard 7.3B parameter model, with optimizations for speed and context length.","cacheWritesPrice":0.059,"cacheReadsPrice":0.059},"novita/deepseek/deepseek-r1-distill-llama-70b":{"maxTokens":0,"contextWindow":32000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7999999999999999,"outputPrice":0.7999999999999999,"description":"DeepSeek R1 Distill LLama 70B","cacheWritesPrice":0.7999999999999999,"cacheReadsPrice":0.7999999999999999},"novita/meta-llama/llama-3.1-8b-instruct-max":{"maxTokens":0,"contextWindow":16384,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.049999999999999996,"outputPrice":0.049999999999999996,"description":"Meta's latest class of models, Llama 3.1, launched with a variety of sizes and configurations. The 8B instruct-tuned version is particularly fast and efficient. It has demonstrated strong performance in human evaluations, outperforming several leading closed-source models.","cacheWritesPrice":0.049999999999999996,"cacheReadsPrice":0.049999999999999996},"novita/meta-llama/llama-3.1-70b-instruct":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.33999999999999997,"outputPrice":0.39,"description":"Meta's latest class of models, Llama 3.1, has launched with a variety of sizes and configurations. The 70B instruct-tuned version is optimized for high-quality dialogue use cases. It has demonstrated strong performance in human evaluations compared to leading closed-source models.","cacheWritesPrice":0.33999999999999997,"cacheReadsPrice":0.33999999999999997},"novita/moonshotai/kimi-k2-instruct":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.5700000000000001,"outputPrice":2.3,"description":"","cacheWritesPrice":0.5700000000000001,"cacheReadsPrice":0.5700000000000001},"novita/sao10k/l3-70b-euryale-v2.1":{"maxTokens":0,"contextWindow":16000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.48,"outputPrice":1.48,"description":"The uncensored llama3 model is a powerhouse of creativity, excelling in both roleplay and story writing. It offers a liberating experience during roleplays, free from any restrictions. This model stands out for its immense creativity, boasting a vast array of unique ideas and plots, truly a treasure trove for those seeking originality. Its unrestricted nature during roleplays allows for the full breadth of imagination to unfold, akin to an enhanced, big-brained version of Stheno. Perfect for creative minds seeking a boundless platform for their imaginative expressions, the uncensored llama3 model is an ideal choice","cacheWritesPrice":1.48,"cacheReadsPrice":1.48},"novita/sao10k/l31-70b-euryale-v2.2":{"maxTokens":0,"contextWindow":16000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.48,"outputPrice":1.48,"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from Sao10k. It is the successor of Euryale L3 70B v2.1.","cacheWritesPrice":1.48,"cacheReadsPrice":1.48},"novita/meta-llama/llama-3.2-3b-instruct":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.03,"outputPrice":0.049999999999999996,"description":"The Meta Llama 3.2 collection of multilingual large language models (LLMs) is a collection of pretrained and instruction-tuned generative models in 1B and 3B sizes (text in/text out)","cacheWritesPrice":0.03,"cacheReadsPrice":0.03},"novita/mistralai/mistral-nemo":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.16999999999999998,"outputPrice":0.16999999999999998,"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese, Korean, Arabic, and Hindi. It supports function calling and is released under the Apache 2.0 license.","cacheWritesPrice":0.16999999999999998,"cacheReadsPrice":0.16999999999999998},"novita/meta-llama/llama-3.3-70b-instruct":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39,"outputPrice":0.39,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model is optimized for multilingual dialogue use cases and outperforms many of the available open source and closed chat models on common industry benchmarks.\n\nSupported languages: English, German, French, Italian, Portuguese, Hindi, Spanish, and Thai.","cacheWritesPrice":0.39,"cacheReadsPrice":0.39},"novita/sao10k/l3-8b-lunaris":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.049999999999999996,"outputPrice":0.049999999999999996,"description":"A generalist / roleplaying model merge based on Llama 3.","cacheWritesPrice":0.049999999999999996,"cacheReadsPrice":0.049999999999999996},"novita/qwen/qwq-32b":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.18,"outputPrice":0.19999999999999998,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.18,"cacheReadsPrice":0.18},"novita/deepseek/deepseek-r1-turbo":{"maxTokens":0,"contextWindow":64000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7,"outputPrice":2.5,"description":"DeepSeek R1 is the latest open-source model released by the DeepSeek team, featuring impressive reasoning capabilities, particularly achieving performance comparable to OpenAI's o1 model in mathematics, coding, and reasoning tasks.","cacheWritesPrice":2.5,"cacheReadsPrice":0.7},"novita/teknium/openhermes-2.5-mistral-7b":{"maxTokens":0,"contextWindow":4096,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.16999999999999998,"outputPrice":0.16999999999999998,"description":"OpenHermes 2.5 Mistral 7B is a state of the art Mistral Fine-tune, a continuation of OpenHermes 2 model, which trained on additional code datasets.","cacheWritesPrice":0.16999999999999998,"cacheReadsPrice":0.16999999999999998},"novita/qwen/qwen-2-vl-72b-instruct":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.44999999999999996,"outputPrice":0.44999999999999996,"description":"Qwen2 VL 72B is a multimodal LLM from the Qwen Team with the following key enhancements:\n\nSoTA understanding of images of various resolution & ratio: Qwen2-VL achieves state-of-the-art performance on visual understanding benchmarks, including MathVista, DocVQA, RealWorldQA, MTVQA, etc.\n\nUnderstanding videos of 20min+: Qwen2-VL can understand videos over 20 minutes for high-quality video-based question answering, dialog, content creation, etc.\n\nAgent that can operate your mobiles, robots, etc.: with the abilities of complex reasoning and decision making, Qwen2-VL can be integrated with devices like mobile phones, robots, etc., for automatic operation based on visual environment and text instructions.\n\nMultilingual Support: to serve global users, besides English and Chinese, Qwen2-VL now supports the understanding of texts in different languages inside images, including most European languages, Japanese, Korean, Arabic, Vietnamese, etc.","cacheWritesPrice":0.44999999999999996,"cacheReadsPrice":0.44999999999999996},"novita/deepseek/deepseek-prover-v2-671b":{"maxTokens":0,"contextWindow":160000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7,"outputPrice":2.5,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":0.7,"cacheReadsPrice":0.7},"novita/deepseek/deepseek-r1-distill-qwen-32b":{"maxTokens":0,"contextWindow":12800,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":0.3,"description":"DeepSeek R1 Distill Qwen 32B is a distilled large language model based on Qwen 2.5 32B, using outputs from DeepSeek R1. It outperforms OpenAI's o1-mini across various benchmarks, achieving new state-of-the-art results for dense models.\n\nOther benchmark results include:\nAIME 2024 pass@1: 72.6\nMATH-500 pass@1: 94.3\nCodeForces Rating: 1691\nThe model leverages fine-tuning from DeepSeek R1's outputs, enabling competitive performance comparable to larger frontier models.","cacheWritesPrice":0.3,"cacheReadsPrice":0.3},"novita/qwen/qwen2.5-vl-72b-instruct":{"maxTokens":0,"contextWindow":96000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7999999999999999,"outputPrice":0.7999999999999999,"description":"Qwen2 VL 72B is a multimodal LLM from the Qwen Team with the following key enhancements:\n\nSoTA understanding of images of various resolution & ratio: Qwen2-VL achieves state-of-the-art performance on visual understanding benchmarks, including MathVista, DocVQA, RealWorldQA, MTVQA, etc.\n\nUnderstanding videos of 20min+: Qwen2-VL can understand videos over 20 minutes for high-quality video-based question answering, dialog, content creation, etc.\n\nAgent that can operate your mobiles, robots, etc.: with the abilities of complex reasoning and decision making, Qwen2-VL can be integrated with devices like mobile phones, robots, etc., for automatic operation based on visual environment and text instructions.\n\nMultilingual Support: to serve global users, besides English and Chinese, Qwen2-VL now supports the understanding of texts in different languages inside images, including most European languages, Japanese, Korean, Arabic, Vietnamese, etc.","cacheWritesPrice":0.7999999999999999,"cacheReadsPrice":0.7999999999999999},"novita/Sao10K/L3-8B-Stheno-v3.2":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.049999999999999996,"outputPrice":0.049999999999999996,"description":"Sao10K/L3-8B-Stheno-v3.2 is a highly skilled actor that excels at fully immersing itself in any role assigned.","cacheWritesPrice":0.049999999999999996,"cacheReadsPrice":0.049999999999999996},"together/meta-llama/Llama-2-70b-hf":{"maxTokens":0,"contextWindow":4096,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.8999999999999999,"outputPrice":0.8999999999999999,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.8999999999999999,"cacheReadsPrice":0.8999999999999999},"together/meta-llama/Meta-Llama-3-8B-Instruct-Lite":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.09999999999999999,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.09999999999999999},"together/nvidia/Llama-3.1-Nemotron-70B-Instruct-HF":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.88,"outputPrice":0.88,"description":"Llama-3.3-Nemotron-Super-49B-v1 is a large language model (LLM) optimized for advanced reasoning, conversational interactions, retrieval-augmented generation (RAG), and tool-calling tasks. Derived from Meta's Llama-3.3-70B-Instruct, it employs a Neural Architecture Search (NAS) approach, significantly enhancing efficiency and reducing memory requirements. This allows the model to support a context length of up to 128K tokens and fit efficiently on single high-performance GPUs, such as NVIDIA H200.\n\nNote: you must include `detailed thinking on` in the system prompt to enable reasoning. Please see [Usage Recommendations](https://huggingface.co/nvidia/Llama-3_1-Nemotron-Ultra-253B-v1#quick-start-and-usage-recommendations) for more.","cacheWritesPrice":0.88,"cacheReadsPrice":0.88},"together/Qwen/Qwen2.5-7B-Instruct-Turbo":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":0.3,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.3,"cacheReadsPrice":0.3},"together/meta-llama/Llama-3.3-70B-Instruct-Turbo":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.88,"outputPrice":0.88,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.88,"cacheReadsPrice":0.88},"together/meta-llama/Llama-2-13b-chat-hf":{"maxTokens":0,"contextWindow":4096,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.22,"outputPrice":0.22,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.22,"cacheReadsPrice":0.22},"together/deepseek-ai/DeepSeek-V3":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":1.25,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":1.25,"cacheReadsPrice":1.25},"together/NousResearch/Nous-Hermes-2-Mixtral-8x7B-DPO":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.6,"outputPrice":0.6,"description":"DeepHermes 3 (Mistral 24B Preview) is an instruction-tuned language model by Nous Research based on Mistral-Small-24B, designed for chat, function calling, and advanced multi-turn reasoning. It introduces a dual-mode system that toggles between intuitive chat responses and structured “deep reasoning” mode using special system prompts. Fine-tuned via distillation from R1, it supports structured output (JSON mode) and function call syntax for agent-based applications.\n\nDeepHermes 3 supports a **reasoning toggle via system prompt**, allowing users to switch between fast, intuitive responses and deliberate, multi-step reasoning. When activated with the following specific system instruction, the model enters a *\"deep thinking\"* mode—generating extended chains of thought wrapped in `<think></think>` tags before delivering a final answer. \n\nSystem Prompt: You are a deep thinking AI, you may use extremely long chains of thought to deeply consider the problem and deliberate with yourself via systematic reasoning processes to help come to a correct solution prior to answering. You should enclose your thoughts and internal monologue inside <think> </think> tags, and then provide your solution or response to the problem.","cacheWritesPrice":0.6,"cacheReadsPrice":0.6},"together/Qwen/Qwen2-72B-Instruct":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.8999999999999999,"outputPrice":0.8999999999999999,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.8999999999999999,"cacheReadsPrice":0.8999999999999999},"together/meta-llama/Meta-Llama-3-70B-Instruct-Lite":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.54,"outputPrice":0.54,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.54,"cacheReadsPrice":0.54},"together/meta-llama/Llama-2-7b-chat-hf":{"maxTokens":0,"contextWindow":4096,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":0.19999999999999998,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.19999999999999998,"cacheReadsPrice":0.19999999999999998},"together/meta-llama/Llama-3.2-3B-Instruct-Turbo":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.06,"outputPrice":0.06,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.06,"cacheReadsPrice":0.06},"together/meta-llama/Llama-3-8b-chat-hf":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":0.19999999999999998,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.19999999999999998,"cacheReadsPrice":0.19999999999999998},"together/upstage/SOLAR-10.7B-Instruct-v1.0":{"maxTokens":0,"contextWindow":4096,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":0.3,"description":"","cacheWritesPrice":0.3,"cacheReadsPrice":0.3},"together/deepseek-ai/DeepSeek-R1":{"maxTokens":8192,"contextWindow":64000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":7,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":7,"cacheReadsPrice":3},"together/Qwen/Qwen2.5-Coder-32B-Instruct":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7999999999999999,"outputPrice":0.7999999999999999,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":0.7999999999999999,"cacheReadsPrice":0.7999999999999999},"together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.88,"outputPrice":0.88,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.88,"cacheReadsPrice":0.88},"together/meta-llama/Meta-Llama-Guard-3-8B":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":0.19999999999999998,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.19999999999999998,"cacheReadsPrice":0.19999999999999998},"together/meta-llama/LlamaGuard-2-8b":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":0.19999999999999998,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.19999999999999998,"cacheReadsPrice":0.19999999999999998},"together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.18,"outputPrice":0.18,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.18,"cacheReadsPrice":0.18},"together/deepseek-llm-67b-chat":{"maxTokens":0,"contextWindow":4096,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.8999999999999999,"outputPrice":0.8999999999999999,"description":"DeepSeek-R1-Distill-Qwen-7B is a 7 billion parameter dense language model distilled from DeepSeek-R1, leveraging reinforcement learning-enhanced reasoning data generated by DeepSeek's larger models. The distillation process transfers advanced reasoning, math, and code capabilities into a smaller, more efficient model architecture based on Qwen2.5-Math-7B. This model demonstrates strong performance across mathematical benchmarks (92.8% pass@1 on MATH-500), coding tasks (Codeforces rating 1189), and general reasoning (49.1% pass@1 on GPQA Diamond), achieving competitive accuracy relative to larger models while maintaining smaller inference costs.","cacheWritesPrice":0.8999999999999999,"cacheReadsPrice":0.8999999999999999},"together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo":{"maxTokens":0,"contextWindow":130815,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":3.5,"outputPrice":3.5,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":3.5,"cacheReadsPrice":3.5},"together/meta-llama/Meta-Llama-3-70B-Instruct-Turbo":{"maxTokens":0,"contextWindow":8192,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.88,"outputPrice":0.88,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.88,"cacheReadsPrice":0.88},"together/Qwen/Qwen2.5-72B-Instruct-Turbo":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.2,"outputPrice":1.2,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":1.2,"cacheReadsPrice":1.2},"together/Qwen/QwQ-32B-Preview":{"maxTokens":0,"contextWindow":32768,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.2,"outputPrice":1.2,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique ability to switch seamlessly between a thinking mode for complex reasoning and a non-thinking mode for efficient dialogue ensures versatile, high-quality performance.\n\nSignificantly outperforming prior models like QwQ and Qwen2.5, Qwen3 delivers superior mathematics, coding, commonsense reasoning, creative writing, and interactive dialogue capabilities. The Qwen3-30B-A3B variant includes 30.5 billion parameters (3.3 billion activated), 48 layers, 128 experts (8 activated per task), and supports up to 131K token contexts with YaRN, setting a new standard among open-source models.","cacheWritesPrice":1.2,"cacheReadsPrice":1.2},"together/meta-llama/Llama-3-70b-chat-hf":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.88,"outputPrice":0.88,"description":"A lightweight and ultra-fast variant of Llama 3.3 70B, for use when quick response times are needed most.","cacheWritesPrice":0.88,"cacheReadsPrice":0.88},"moonshot/kimi-k2-turbo-preview":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1.2,"outputPrice":5,"description":"A Mixture-of-Experts (MoE) foundation model with exceptional coding and agent capabilities, featuring 1 trillion total parameters and 32 billion activated parameters. In benchmark evaluations covering general knowledge reasoning, programming, mathematics, and agent-related tasks, the K2 model outperforms other leading open-source models.","cacheWritesPrice":5,"cacheReadsPrice":0.3},"moonshot/kimi-k2-0711-preview":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.6,"outputPrice":2.5,"description":"A Mixture-of-Experts (MoE) foundation model with exceptional coding and agent capabilities, featuring 1 trillion total parameters and 32 billion activated parameters. In benchmark evaluations covering general knowledge reasoning, programming, mathematics, and agent-related tasks, the K2 model outperforms other leading open-source models.","cacheWritesPrice":0.6,"cacheReadsPrice":0.15},"anthropic/claude-sonnet-4":{"maxTokens":64000,"contextWindow":1000000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Claude Sonnet 4 significantly improves on Sonnet 3.7's industry-leading capabilities, excelling in coding with a state-of-the-art 72.7% on SWE-bench. The model balances performance and efficiency for internal and external use cases, with enhanced steerability for greater control over implementations.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"anthropic/claude-3-5-haiku":{"maxTokens":8192,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.7999999999999999,"outputPrice":4,"description":"Anthropic's fastest model. Intelligence at blazing speeds.","cacheWritesPrice":1,"cacheReadsPrice":0.08},"anthropic/claude-opus-4":{"maxTokens":32000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Claude Opus 4 is Anthropic's most powerful model yet and the best coding model in the world, leading on SWE-bench (72.5%) and Terminal-bench (43.2%). It delivers sustained performance on long-running tasks that require focused effort and thousands of steps, with the ability to work continuously for several hours—dramatically outperforming all Sonnet models and significantly expanding what AI agents can accomplish.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"anthropic/claude-3-7-sonnet":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"anthropic/claude-opus-4-1":{"maxTokens":32000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Claude Opus 4 is Anthropic's most powerful model yet and the best coding model in the world, leading on SWE-bench (72.5%) and Terminal-bench (43.2%). It delivers sustained performance on long-running tasks that require focused effort and thousands of steps, with the ability to work continuously for several hours—dramatically outperforming all Sonnet models and significantly expanding what AI agents can accomplish.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"anthropic/claude-3-opus":{"maxTokens":4096,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Powerful model for highly complex tasks. Top-level intelligence, fluency, and understanding.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"anthropic/claude-haiku-4-5":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":1,"outputPrice":5,"description":"Anthropic Haiku 4.5","cacheWritesPrice":1.25,"cacheReadsPrice":0.09999999999999999},"anthropic/claude-sonnet-4-5":{"maxTokens":64000,"contextWindow":1000000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Claude Sonnet 4.5 is Anthropic's best coding model in the world, leading on SWE-bench Verified (77.2%) and OSWorld (61.4%). It delivers sustained autonomous performance on complex tasks for over 30 hours—up from seven hours for Opus 4—maintaining focus and reliability throughout the entire software development lifecycle, with enhanced capabilities in tool handling, memory management, and context processing that make it the strongest model for building complex agents.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"anthropic/claude-3-haiku":{"maxTokens":4096,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.25,"outputPrice":1.25,"description":"Fastest and most compact model for near-instant responsiveness. Quick and accurate targeted performance.","cacheWritesPrice":0.3,"cacheReadsPrice":0.03},"smart/task":{"maxTokens":0,"contextWindow":0,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"description":""},"zai/GLM-4.6":{"maxTokens":128000,"contextWindow":200000,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.6,"outputPrice":2.2,"description":"GLM-4.6 is Z AI’s latest flagship model, designed to push agentic and coding performance further. It expands the context window from 128K to 200K tokens, improves reasoning and tool-use capabilities, and delivers stronger results in coding benchmarks and real-world development workflows. GLM-4.6 demonstrates refined writing quality, more capable agent behavior, and higher token efficiency (≈15% fewer tokens vs. GLM-4.5).\n\nEvaluations show clear gains over GLM-4.5 across reasoning, agents, and coding, reaching near parity with Claude Sonnet 4 in practical tasks while outperforming other open-source baselines. GLM-4.6 is available through the Z.ai API platform, OpenRouter, coding agents (Claude Code, Roo Code, Cline, Kilo Code), and soon as downloadable weights on HuggingFace and ModelScope.","cacheWritesPrice":0.6,"cacheReadsPrice":0.11},"zai/GLM-4.5":{"maxTokens":98304,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.6,"outputPrice":2.2,"description":"GLM-4.5 and GLM-4.5-Air are Z AI's latest flagship models, purpose-built as foundational models for agent-oriented applications. Both leverage a Mixture-of-Experts (MoE) architecture. GLM-4.5 has a total parameter count of 355B with 32B active parameters per forward pass, while GLM-4.5-Air adopts a more streamlined design with 106B total parameters and 12B active parameters.","cacheWritesPrice":0.6,"cacheReadsPrice":0.11},"xai/grok-3-mini:low":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":0.5,"description":"A lightweight model that thinks before responding. Fast, smart, and great for logic-based tasks that do not require deep domain knowledge. The raw thinking traces are accessible.","cacheWritesPrice":0.3,"cacheReadsPrice":0.3},"xai/grok-4-fast-non-reasoning":{"maxTokens":0,"contextWindow":2000000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":0.5,"description":"xAI's latest advancement in cost-efficient reasoning models","cacheWritesPrice":0.19999999999999998,"cacheReadsPrice":0.049999999999999996},"xai/grok-2-1212":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":10,"description":"x AI's Our previous generation chat model.\n","cacheWritesPrice":2,"cacheReadsPrice":2},"xai/grok-3-mini":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":0.5,"description":"A lightweight model that thinks before responding. Fast, smart, and great for logic-based tasks that do not require deep domain knowledge. The raw thinking traces are accessible.","cacheWritesPrice":0.3,"cacheReadsPrice":0.3},"xai/grok-3-mini:high":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":0.5,"description":"A lightweight model that thinks before responding. Fast, smart, and great for logic-based tasks that do not require deep domain knowledge. The raw thinking traces are accessible.","cacheWritesPrice":0.3,"cacheReadsPrice":0.3},"xai/grok-code-fast-1":{"maxTokens":0,"contextWindow":256000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":1.5,"description":"A speedy and economical reasoning model that excels at agentic coding","cacheWritesPrice":0.19999999999999998,"cacheReadsPrice":0.02},"xai/grok-3":{"maxTokens":0,"contextWindow":131072,"supportsPromptCache":false,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":5,"outputPrice":25,"description":"Excels at enterprise use cases like data extraction, coding, and text summarization. Possesses deep domain knowledge in finance, healthcare, law, and science.","cacheWritesPrice":5,"cacheReadsPrice":5},"xai/grok-4":{"maxTokens":0,"contextWindow":256000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"xAI's latest and greatest flagship model, offering unparalleled performance in natural language, math and reasoning - the perfect jack of all trades.","cacheWritesPrice":3,"cacheReadsPrice":0.75},"xai/grok-4-fast":{"maxTokens":0,"contextWindow":2000000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.19999999999999998,"outputPrice":0.5,"description":"xAI's latest advancement in cost-efficient reasoning models","cacheWritesPrice":0.19999999999999998,"cacheReadsPrice":0.049999999999999996},"openai/gpt-4.1-mini":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.39999999999999997,"outputPrice":1.5999999999999999,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard instruction evals, 35.8% on MultiChallenge, and 84.1% on IFEval. Mini also shows strong coding ability (e.g., 31.6% on Aider’s polyglot diff benchmark) and vision understanding, making it suitable for interactive applications with tight performance constraints.","cacheWritesPrice":0.39999999999999997,"cacheReadsPrice":0.09999999999999999},"openai/o4-mini:high":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.275},"openai/o1:low":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":15,"outputPrice":60,"description":"The o1 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","cacheWritesPrice":15,"cacheReadsPrice":7.5},"openai/o3-mini:medium":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.55},"openai/gpt-5-nano":{"maxTokens":128000,"contextWindow":400000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.049999999999999996,"outputPrice":0.39999999999999997,"description":"GPT-5 nano is OpenAI's fastest, cheapest version of GPT-5. It's great for summarization and classification tasks.","cacheWritesPrice":0.049999999999999996,"cacheReadsPrice":0.005},"openai/gpt-5":{"maxTokens":128000,"contextWindow":400000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"GPT-5 is OpenAI's flagship model for coding, reasoning, and agentic tasks across domains.","cacheWritesPrice":10,"cacheReadsPrice":0.125},"openai/o4-mini:low":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.275},"openai/o3-mini:high":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.55},"openai/gpt-5-mini:priority":{"maxTokens":128000,"contextWindow":400000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.44999999999999996,"outputPrice":3.5999999999999996,"description":"GPT-5 is OpenAI's flagship model for coding, reasoning, and agentic tasks across domains.","cacheWritesPrice":3.5999999999999996,"cacheReadsPrice":0.045},"openai/gpt-4o-2024-05-13":{"maxTokens":4096,"contextWindow":128000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2.5,"outputPrice":10,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded files, providing deeper insights & more thorough responses.\n\nGPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.","cacheWritesPrice":2.5,"cacheReadsPrice":2.5},"openai/o4-mini":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.275},"openai/o1":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":15,"outputPrice":60,"description":"The o1 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","cacheWritesPrice":15,"cacheReadsPrice":7.5},"openai/o3-pro":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":false,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":20,"outputPrice":80,"description":"The o3 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","cacheWritesPrice":20,"cacheReadsPrice":20},"openai/o4-mini-deep-research":{"maxTokens":200000,"contextWindow":100000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":2,"outputPrice":8,"description":"Optimized for fast reasoning and minimal latency, O4 Mini Deep Research supports caching and high-speed inference. Ideal for lightweight agent use.","cacheWritesPrice":2,"cacheReadsPrice":0.5},"openai/gpt-4.1-nano":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.09999999999999999,"outputPrice":0.39999999999999997,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million token context window, and scores 80.1% on MMLU, 50.3% on GPQA, and 9.8% on Aider polyglot coding – even higher than GPT‑4o mini. It’s ideal for tasks like classification or autocompletion.","cacheWritesPrice":0.09999999999999999,"cacheReadsPrice":0.024999999999999998},"openai/o4-mini:flex":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.55,"outputPrice":2.2,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":0.55,"cacheReadsPrice":0.13799999999999998},"openai/gpt-4o-mini":{"maxTokens":16384,"contextWindow":128000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.15,"outputPrice":0.6,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs.\n\nAs their most advanced small model, it is many multiples more affordable than other recent frontier models, and more than 60% cheaper than [GPT-3.5 Turbo](/models/openai/gpt-3.5-turbo). It maintains SOTA intelligence, while being significantly more cost-effective.\n\nGPT-4o mini achieves an 82% score on MMLU and presently ranks higher than GPT-4 on chat preferences [common leaderboards](https://arena.lmsys.org/).\n\nCheck out the [launch announcement](https://openai.com/index/gpt-4o-mini-advancing-cost-efficient-intelligence/) to learn more.\n\n#multimodal","cacheWritesPrice":0.15,"cacheReadsPrice":0.075},"openai/o1-mini":{"maxTokens":65536,"contextWindow":128000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.1,"outputPrice":4.4,"description":"The o1 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023. o1-mini is a faster and more affordable reasoning model, but OpenAI recommends using the newer o3-mini model that features higher intelligence at the same latency and price as o1-mini.","cacheWritesPrice":1.1,"cacheReadsPrice":0.55},"openai/o3-deep-research":{"maxTokens":200000,"contextWindow":100000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":10,"outputPrice":40,"description":"O3 Deep Research is a premium OpenAI model tuned for long-context research and high-recall reasoning tasks, optimized for analytical depth.","cacheWritesPrice":10,"cacheReadsPrice":2.5},"openai/gpt-5-mini:flex":{"maxTokens":128000,"contextWindow":400000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.125,"outputPrice":1,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","cacheWritesPrice":0.125,"cacheReadsPrice":0.012499999999999999},"openai/o1-pro":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":false,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":150,"outputPrice":600,"description":"The o1 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","cacheWritesPrice":150,"cacheReadsPrice":150},"openai/gpt-5:priority":{"maxTokens":128000,"contextWindow":400000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":2.5,"outputPrice":20,"description":"GPT-5 is OpenAI's flagship model for coding, reasoning, and agentic tasks across domains.","cacheWritesPrice":20,"cacheReadsPrice":0.25},"openai/gpt-4o-2024-11-20":{"maxTokens":16384,"contextWindow":128000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2.5,"outputPrice":10,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded files, providing deeper insights & more thorough responses.\n\nGPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.","cacheWritesPrice":2.5,"cacheReadsPrice":1.25},"openai/gpt-5-mini":{"maxTokens":128000,"contextWindow":400000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.25,"outputPrice":2,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","cacheWritesPrice":0.25,"cacheReadsPrice":0.024999999999999998},"openai/gpt-4.1":{"maxTokens":32768,"contextWindow":1047576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2,"outputPrice":8,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and GPT-4.5 across coding (54.6% SWE-bench Verified), instruction compliance (87.4% IFEval), and multimodal understanding benchmarks. It is tuned for precise code diffs, agent reliability, and high recall in large document contexts, making it ideal for agents, IDE tooling, and enterprise knowledge retrieval.","cacheWritesPrice":2,"cacheReadsPrice":0.5},"openai/o3-mini:low":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.55},"openai/chatgpt-4o":{"maxTokens":16000,"contextWindow":128000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":5,"outputPrice":15,"description":"OpenAI ChatGPT 4o is continually updated by OpenAI to point to the current version of GPT-4o used by ChatGPT. It therefore differs slightly from the API version of [GPT-4o](/models/openai/gpt-4o) in that it has additional RLHF. It is intended for research and evaluation.\n\nOpenAI notes that this model is not suited for production use-cases as it may be removed or redirected to another model in the future.","cacheWritesPrice":5,"cacheReadsPrice":5},"openai/gpt-4o":{"maxTokens":16384,"contextWindow":128000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2.5,"outputPrice":10,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded files, providing deeper insights & more thorough responses.\n\nGPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.","cacheWritesPrice":2.5,"cacheReadsPrice":1.25},"openai/o4-mini:medium":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.275},"openai/o1:high":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":15,"outputPrice":60,"description":"The o1 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","cacheWritesPrice":15,"cacheReadsPrice":7.5},"openai/gpt-5:flex":{"maxTokens":128000,"contextWindow":400000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.625,"outputPrice":5,"description":"GPT-5 is OpenAI's flagship model for coding, reasoning, and agentic tasks across domains.","cacheWritesPrice":0.625,"cacheReadsPrice":0.0625},"openai/gpt-4o-2024-08-06":{"maxTokens":16384,"contextWindow":128000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":2.5,"outputPrice":10,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded files, providing deeper insights & more thorough responses.\n\nGPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as fast and 50% more cost-effective. GPT-4o also offers improved performance in processing non-English languages and enhanced visual capabilities.","cacheWritesPrice":2.5,"cacheReadsPrice":1.25},"openai/o1:medium":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":15,"outputPrice":60,"description":"The o1 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","cacheWritesPrice":15,"cacheReadsPrice":7.5},"openai/gpt-5-chat":{"maxTokens":16384,"contextWindow":128000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"GPT-5 is OpenAI's flagship model for coding, reasoning, and agentic tasks across domains.","cacheWritesPrice":10,"cacheReadsPrice":0.125},"openai/gpt-5-nano:flex":{"maxTokens":128000,"contextWindow":400000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.024999999999999998,"outputPrice":0.19999999999999998,"description":"GPT-5 nano is OpenAI's fastest, cheapest version of GPT-5. It's great for summarization and classification tasks.","cacheWritesPrice":0.024999999999999998,"cacheReadsPrice":0.0025},"openai/o3:flex":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1,"outputPrice":4,"description":"O3 Flex is a cheaper version of the o3 model","cacheWritesPrice":1,"cacheReadsPrice":0.25},"openai/o3-mini":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.55},"openai/o3":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":2,"outputPrice":8,"description":"The o1 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","cacheWritesPrice":2,"cacheReadsPrice":0.5},"openai-responses/o4-mini":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.275},"openai-responses/gpt-5-pro":{"maxTokens":272000,"contextWindow":400000,"supportsPromptCache":false,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":15,"outputPrice":120,"description":"GPT-5 Pro is OpenAI’s extended-reasoning tier of GPT-5, built to push reliability on hard problems, long tool chains, and agentic workflows. It keeps GPT-5’s multimodal skills and very large context (API page lists up to 400K tokens) while allocating more compute to think longer and plan better, improving code generation, math, and complex writing beyond standard GPT-5/“Thinking.” OpenAI positions Pro as the version that “uses extended reasoning for even more comprehensive and accurate answers,” targeting high-stakes tasks and enterprise use.","cacheWritesPrice":120,"cacheReadsPrice":15},"openai-responses/gpt-5-codex":{"maxTokens":128000,"contextWindow":400000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.25,"outputPrice":10,"description":"GPT-5-Codex is a version of GPT-5 optimized for agentic coding tasks in Codex or similar environments","cacheWritesPrice":1.25,"cacheReadsPrice":0.125},"openai-responses/gpt-5-mini":{"maxTokens":128000,"contextWindow":400000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.25,"outputPrice":2,"description":"GPT-5 mini is a faster, more cost-efficient version of GPT-5. It's great for well-defined tasks and precise prompts.","cacheWritesPrice":0.25,"cacheReadsPrice":0.024999999999999998},"openai-responses/o3-mini":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":1.1,"outputPrice":4.4,"description":"o3-mini is OpenAI's most recent small reasoning model, providing high intelligence at the same cost and latency targets of o1-mini. o3-mini also supports key developer features, like Structured Outputs, function calling, Batch API, and more. Like other models in the o-series, it is designed to excel at science, math, and coding tasks.","cacheWritesPrice":1.1,"cacheReadsPrice":0.55},"openai-responses/o3-pro":{"maxTokens":100000,"contextWindow":200000,"supportsPromptCache":false,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":20,"outputPrice":80,"description":"The o3 series of models are trained with reinforcement learning to perform complex reasoning. o1 models think before they answer, producing a long internal chain of thought before responding to the user. The o1 reasoning model is designed to solve hard problems across domains. The knowledge cutoff for o1 and o1-mini models is October, 2023.","cacheWritesPrice":20,"cacheReadsPrice":20},"openai-responses/gpt-5-nano":{"maxTokens":128000,"contextWindow":400000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":false,"supportsReasoningEffort":true,"inputPrice":0.049999999999999996,"outputPrice":0.39999999999999997,"description":"GPT-5 nano is OpenAI's fastest, cheapest version of GPT-5. It's great for summarization and classification tasks.","cacheWritesPrice":0.049999999999999996,"cacheReadsPrice":0.005},"coding/gemini-2.5-pro@europe-central2":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-flash@europe-west1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"coding/gemini-2.5-pro":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-pro@us-central1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-pro@us-south1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-flash@us-east5":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"coding/gemini-2.5-flash@europe-central2":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"coding/gemini-2.5-pro@europe-west8":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-pro-preview-03-25":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-pro@us-east1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-pro@us-east5":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-pro@europe-west1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-flash@us-east1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"coding/gemini-2.5-flash@europe-north1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"coding/gemini-2.5-flash@europe-west8":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"coding/claude-3-7-sonnet-20250219:8192":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"coding/claude-3-7-sonnet-20250219:16384":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"coding/claude-3-7-sonnet-20250219:high":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"coding/claude-3-7-sonnet-20250219:max":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"coding/claude-sonnet-4-20250514":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Claude Sonnet 4 significantly improves on Sonnet 3.7's industry-leading capabilities, excelling in coding with a state-of-the-art 72.7% on SWE-bench. The model balances performance and efficiency for internal and external use cases, with enhanced steerability for greater control over implementations.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"coding/gemini-2.5-flash-preview-05-20":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.15,"outputPrice":0.6,"description":"","cacheWritesPrice":0.39999999999999997,"cacheReadsPrice":0.15},"coding/gemini-2.5-pro@us-west1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-pro@europe-west4":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-flash":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"coding/gemini-2.5-flash@us-west1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"coding/claude-opus-4-20250514":{"maxTokens":32000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":15,"outputPrice":75,"description":"Claude Opus 4 is Anthropic's most powerful model yet and the best coding model in the world, leading on SWE-bench (72.5%) and Terminal-bench (43.2%). It delivers sustained performance on long-running tasks that require focused effort and thousands of steps, with the ability to work continuously for several hours—dramatically outperforming all Sonnet models and significantly expanding what AI agents can accomplish.","cacheWritesPrice":18.75,"cacheReadsPrice":1.5},"coding/claude-3-7-sonnet-20250219:64000":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"coding/gemini-2.5-pro-preview-05-06":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-flash@us-south1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"coding/claude-3-7-sonnet-20250219":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"coding/claude-3-7-sonnet-20250219:1024":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"coding/claude-3-7-sonnet-20250219:low":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"coding/claude-3-7-sonnet-20250219:medium":{"maxTokens":64000,"contextWindow":200000,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":3,"outputPrice":15,"description":"Anthropic's most intelligent model. The first hybrid reasoning model on the market with the highest level of intelligence and capability with toggleable extended thinking. Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing.","cacheWritesPrice":3.75,"cacheReadsPrice":0.3},"coding/gemini-2.5-pro@europe-north1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":1.25,"outputPrice":10,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy and nuanced context handling. Gemini 2.5 Pro achieves top-tier performance on multiple benchmarks, including first-place positioning on the LMArena leaderboard, reflecting superior human-preference alignment and complex problem-solving abilities.","cacheWritesPrice":2.375,"cacheReadsPrice":0.31},"coding/gemini-2.5-flash@us-central1":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"coding/gemini-2.5-flash@europe-west4":{"maxTokens":65535,"contextWindow":1048576,"supportsPromptCache":true,"supportsImages":true,"supportsReasoningBudget":true,"supportsReasoningEffort":false,"inputPrice":0.3,"outputPrice":2.5,"description":"Google's first hybrid reasoning model which supports a 1M token context window and has thinking budgets. Most balanced Gemini model, optimized for low latency use cases.","cacheWritesPrice":0.55,"cacheReadsPrice":0.075},"deepseek/deepseek-chat":{"maxTokens":8000,"contextWindow":128000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.28,"outputPrice":0.42,"description":"DeepSeek-V3 achieves a significant breakthrough in inference speed over previous models.\n\nIt tops the leaderboard among open-source models and rivals the most advanced closed-source models globally.","cacheWritesPrice":0.42,"cacheReadsPrice":0.028},"deepseek/deepseek-reasoner":{"maxTokens":64000,"contextWindow":128000,"supportsPromptCache":true,"supportsImages":false,"supportsReasoningBudget":false,"supportsReasoningEffort":false,"inputPrice":0.28,"outputPrice":0.42,"description":"Fully open-source model & technical report. Performance on par with OpenAI-o1.","cacheWritesPrice":0.28,"cacheReadsPrice":0.028}}