{"fireworks-ai": {"id": "fireworks-ai", "env": ["FIREWORKS_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.fireworks.ai/inference/v1/", "name": "Fireworks AI", "doc": "https://fireworks.ai/docs/", "models": {"accounts/fireworks/routers/glm-5p2-fast": {"id": "accounts/fireworks/routers/glm-5p2-fast", "name": "GLM 5.2 Fast", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-06-26", "last_updated": "2026-06-26", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048575, "output": 131072}, "cost": {"input": 2.1, "output": 6.6, "cache_read": 0.21}}, "accounts/fireworks/routers/kimi-k2p6-fast": {"id": "accounts/fireworks/routers/kimi-k2p6-fast", "name": "Kimi K2.6 Fast", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-06-05", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262000, "output": 262000}, "cost": {"input": 2, "output": 8, "cache_read": 0.3}}, "accounts/fireworks/routers/kimi-k2p6-turbo": {"id": "accounts/fireworks/routers/kimi-k2p6-turbo", "name": "Kimi K2.6 Turbo", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262000, "output": 262000}, "cost": {"input": 2, "output": 8, "cache_read": 0.3}}, "accounts/fireworks/routers/kimi-k2p7-code-fast": {"id": "accounts/fireworks/routers/kimi-k2p7-code-fast", "name": "Kimi K2.7 Code Fast", "description": "Kimi coding model for software agents, refactors, and repository reasoning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-06-12", "last_updated": "2026-06-16", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262000, "output": 262000}, "cost": {"input": 1.9, "output": 8, "cache_read": 0.38}}, "accounts/fireworks/routers/kimi-k3-fast": {"id": "accounts/fireworks/routers/kimi-k3-fast", "name": "Kimi K3 Fast", "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work", "family": "kimi-k3", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "max"]}, {"type": "budget_tokens", "min": 1024}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": false, "release_date": "2026-07-27", "last_updated": "2026-07-27", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 131072}, "cost": {"input": 4.5, "output": 22.5, "cache_read": 0.45}}, "accounts/fireworks/models/qwen3p7-plus": {"id": "accounts/fireworks/models/qwen3p7-plus", "name": "Qwen 3.7 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high"]}, {"type": "budget_tokens", "min": 1}], "tool_call": true, "temperature": true, "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.4, "output": 1.6, "cache_read": 0.08}}, "accounts/fireworks/models/deepseek-v4-flash": {"id": "accounts/fireworks/models/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-06-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 384000}, "cost": {"input": 0.14, "output": 0.28, "cache_read": 0.028}}, "accounts/fireworks/models/gpt-oss-20b": {"id": "accounts/fireworks/models/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.07, "output": 0.3, "cache_read": 0.035}}, "accounts/fireworks/models/minimax-m2p7": {"id": "accounts/fireworks/models/minimax-m2p7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-04-12", "last_updated": "2026-04-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 196608, "output": 196608}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.06}}, "accounts/fireworks/models/kimi-k2p6": {"id": "accounts/fireworks/models/kimi-k2p6", "name": "Kimi K2.6", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262000, "output": 262000}, "cost": {"input": 0.95, "output": 4, "cache_read": 0.16}}, "accounts/fireworks/models/minimax-m3": {"id": "accounts/fireworks/models/minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 512000, "output": 512000}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.06}}, "accounts/fireworks/models/deepseek-v4-pro": {"id": "accounts/fireworks/models/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 384000}, "cost": {"input": 1.74, "output": 3.48, "cache_read": 0.145}}, "accounts/fireworks/models/deepseek-v4-flash-0731": {"id": "accounts/fireworks/models/deepseek-v4-flash-0731", "name": "DeepSeek V4 Flash 0731", "description": "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-07-31", "last_updated": "2026-07-31", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 384000}, "cost": {"input": 0.14, "output": 0.28, "cache_read": 0.028}}, "accounts/fireworks/models/kimi-k2p7-code": {"id": "accounts/fireworks/models/kimi-k2p7-code", "name": "Kimi K2.7 Code", "description": "Kimi coding model for software agents, refactors, and repository reasoning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-06-12", "last_updated": "2026-06-16", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262000, "output": 262000}, "cost": {"input": 0.95, "output": 4, "cache_read": 0.19}}, "accounts/fireworks/models/gpt-oss-120b": {"id": "accounts/fireworks/models/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2026-06-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.15, "output": 0.6, "cache_read": 0.015}}, "accounts/fireworks/models/glm-5p2": {"id": "accounts/fireworks/models/glm-5p2", "name": "GLM 5.2", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-06-16", "last_updated": "2026-06-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048575, "output": 131072}, "cost": {"input": 1.4, "output": 4.4, "cache_read": 0.14}}, "accounts/fireworks/models/kimi-k3": {"id": "accounts/fireworks/models/kimi-k3", "name": "Kimi K3", "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work", "family": "kimi-k3", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "max"]}, {"type": "budget_tokens", "min": 1024}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": false, "release_date": "2026-07-27", "last_updated": "2026-07-27", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 131072}, "cost": {"input": 3, "output": 15, "cache_read": 0.3}}, "fireworks/flux-kontext-pro": {"name": "FLUX.1 Kontext Pro", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.04, "type": "request"}, "id": "fireworks/flux-kontext-pro"}, "fireworks/flux-kontext-max": {"name": "FLUX.1 Kontext Max", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.08, "type": "request"}, "id": "fireworks/flux-kontext-max"}, "fireworks/flux-1-dev-fp8": {"name": "FLUX.1 Dev FP8", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.0005, "type": "per_step"}, "id": "fireworks/flux-1-dev-fp8"}, "fireworks/flux-1-schnell-fp8": {"name": "FLUX.1 [schnell] FP8", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.00035, "type": "per_step"}, "id": "fireworks/flux-1-schnell-fp8"}}}, "nvidia": {"id": "nvidia", "env": ["NVIDIA_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://integrate.api.nvidia.com/v1", "name": "Nvidia", "doc": "https://docs.api.nvidia.com/nim/", "models": {"microsoft/phi-4-mini-instruct": {"id": "microsoft/phi-4-mini-instruct", "name": "Phi-4-Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "phi", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2024-12-01", "last_updated": "2025-09-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0, "output": 0}}, "microsoft/phi-4-multimodal-instruct": {"id": "microsoft/phi-4-multimodal-instruct", "name": "Phi 4 Multimodal", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-07-26", "last_updated": "2025-07-26", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "input": 128000, "output": 16384}, "cost": {"input": 0, "output": 0}}, "nvidia/magpie-tts-zeroshot": {"id": "nvidia/magpie-tts-zeroshot", "name": "magpie-tts-zeroshot", "description": "Speech generation model for controllable voice, narration, and audio delivery", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-05-22", "last_updated": "2025-06-12", "modalities": {"input": ["text", "audio"], "output": ["audio"]}, "open_weights": true, "limit": {"context": 0, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/nv-embedcode-7b-v1": {"id": "nvidia/nv-embedcode-7b-v1", "name": "nv-embedcode-7b-v1", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-03-17", "last_updated": "2025-05-29", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 2048}, "cost": {"input": 0, "output": 0}}, "nvidia/studiovoice": {"id": "nvidia/studiovoice", "name": "studiovoice", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-03", "last_updated": "2025-06-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "nvidia/sparsedrive": {"id": "nvidia/sparsedrive", "name": "sparsedrive", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-18", "last_updated": "2025-07-20", "modalities": {"input": ["video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "nvidia/cosmos-reason2-8b": {"id": "nvidia/cosmos-reason2-8b", "name": "Cosmos Reason2 8B", "description": "Vision language model for physical-world understanding with structured reasoning on video and images", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-nano-12b-v2-vl": {"id": "nvidia/nemotron-nano-12b-v2-vl", "name": "Nemotron Nano 12B v2 VL", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-28", "last_updated": "2025-10-28", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 0, "output": 0}}, "nvidia/bevformer": {"id": "nvidia/bevformer", "name": "bevformer", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-03-18", "last_updated": "2025-07-20", "modalities": {"input": ["video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3-nano-30b-a3b": {"id": "nvidia/nemotron-3-nano-30b-a3b", "name": "nemotron-3-nano-30b-a3b", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-12", "last_updated": "2024-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0, "output": 0}}, "nvidia/llama-3.3-nemotron-super-49b-v1.5": {"id": "nvidia/llama-3.3-nemotron-super-49b-v1.5", "name": "Llama 3.3 Nemotron Super 49B v1.5", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 0, "output": 0}}, "nvidia/cosmos-transfer2_5-2b": {"id": "nvidia/cosmos-transfer2_5-2b", "name": "cosmos-transfer2.5-2b", "description": "Video model for prompt-guided generation, editing, and motion workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": {"input": ["text", "image", "video"], "output": ["video"]}, "open_weights": true, "limit": {"context": 0, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/llama-3.1-nemotron-nano-8b-v1": {"id": "nvidia/llama-3.1-nemotron-nano-8b-v1", "name": "Llama 3.1 Nemotron Nano 8B v1", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "release_date": "2025-03-18", "last_updated": "2025-03-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0, "output": 0}}, "nvidia/active-speaker-detection": {"id": "nvidia/active-speaker-detection", "name": "Active Speaker Detection", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": {"input": ["video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 0, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/usdvalidate": {"id": "nvidia/usdvalidate", "name": "usdvalidate", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-07-24", "last_updated": "2025-01-08", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 0, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/llama-3.1-nemotron-safety-guard-8b-v3": {"id": "nvidia/llama-3.1-nemotron-safety-guard-8b-v3", "name": "llama-3.1-nemotron-safety-guard-8b-v3", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-10-28", "last_updated": "2025-10-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/llama-3.3-nemotron-super-49b-v1": {"id": "nvidia/llama-3.3-nemotron-super-49b-v1", "name": "Llama 3.3 Nemotron Super 49B v1", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "release_date": "2025-04-07", "last_updated": "2025-04-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 0, "output": 0}}, "nvidia/llama-3.1-nemotron-70b-instruct": {"id": "nvidia/llama-3.1-nemotron-70b-instruct", "name": "Llama 3.1 Nemotron 70B Instruct", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-04-15", "last_updated": "2025-04-15", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3-super-120b-a12b": {"id": "nvidia/nemotron-3-super-120b-a12b", "name": "Nemotron 3 Super", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.2, "output": 0.8}}, "nvidia/nemotron-3-content-safety": {"id": "nvidia/nemotron-3-content-safety", "name": "nemotron-3-content-safety", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/cosmos-transfer1-7b": {"id": "nvidia/cosmos-transfer1-7b", "name": "cosmos-transfer1-7b", "description": "Video model for prompt-guided generation, editing, and motion workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-06-13", "last_updated": "2025-06-30", "modalities": {"input": ["text", "image", "video"], "output": ["video"]}, "open_weights": true, "limit": {"context": 0, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/llama-3.1-nemotron-ultra-253b-v1": {"id": "nvidia/llama-3.1-nemotron-ultra-253b-v1", "name": "Llama 3.1 Nemotron Ultra 253B", "description": "Flagship Nemotron model for high-throughput reasoning and complex agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "release_date": "2025-04-07", "last_updated": "2025-04-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 0, "output": 0}}, "nvidia/llama-3_2-nemoretriever-300m-embed-v1": {"id": "nvidia/llama-3_2-nemoretriever-300m-embed-v1", "name": "llama-3_2-nemoretriever-300m-embed-v1", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-07-24", "last_updated": "2025-07-24", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 2048}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {"id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "name": "Nemotron 3 Nano Omni", "description": "Open Nemotron omni model combining reasoning with text, vision, and audio", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": -1, "max": 32768}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-04-28", "modalities": {"input": ["text", "image", "video", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 65536}, "cost": {"input": 0, "output": 0}}, "nvidia/gliner-pii": {"id": "nvidia/gliner-pii", "name": "gliner-pii", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/rerank-qa-mistral-4b": {"id": "nvidia/rerank-qa-mistral-4b", "name": "rerank-qa-mistral-4b", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-03-17", "last_updated": "2025-01-17", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/llama-3.1-nemotron-nano-vl-8b-v1": {"id": "nvidia/llama-3.1-nemotron-nano-vl-8b-v1", "name": "Llama 3.1 Nemotron Nano VL 8B v1", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-04-10", "last_updated": "2025-04-10", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 16384}, "cost": {"input": 0, "output": 0}}, "nvidia/synthetic-video-detector": {"id": "nvidia/synthetic-video-detector", "name": "synthetic-video-detector", "description": "Video model for prompt-guided generation, editing, and motion workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": {"input": ["video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 0, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/streampetr": {"id": "nvidia/streampetr", "name": "streampetr", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": {"input": ["video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-voicechat": {"id": "nvidia/nemotron-voicechat", "name": "nemotron-voicechat", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "family": "nemotron", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": {"input": ["text", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "nvidia/llama-nemotron-embed-vl-1b-v2": {"id": "nvidia/llama-nemotron-embed-vl-1b-v2", "name": "llama-nemotron-embed-vl-1b-v2", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "nemotron", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-02-10", "last_updated": "2026-02-10", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 2048}, "cost": {"input": 0, "output": 0}}, "nvidia/nvidia-nemotron-nano-9b-v2": {"id": "nvidia/nvidia-nemotron-nano-9b-v2", "name": "nvidia-nemotron-nano-9b-v2", "description": "Compact Nemotron model for efficient reasoning and deployable AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2025-08-18", "last_updated": "2025-08-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0, "output": 0}}, "nvidia/riva-translate-4b-instruct-v1.1": {"id": "nvidia/riva-translate-4b-instruct-v1.1", "name": "riva-translate-4b-instruct-v1_1", "description": "Translation model for multilingual conversion, localization, and cross-language workflows", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-12-12", "last_updated": "2025-12-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-content-safety-reasoning-4b": {"id": "nvidia/nemotron-content-safety-reasoning-4b", "name": "nemotron-content-safety-reasoning-4b", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": false, "release_date": "2026-01-22", "last_updated": "2026-01-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3-ultra-550b-a55b": {"id": "nvidia/nemotron-3-ultra-550b-a55b", "name": "Nemotron 3 Ultra 550B A55B", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.5, "output": 2.5, "cache_read": 0.15}}, "nvidia/llama-nemotron-rerank-vl-1b-v2": {"id": "nvidia/llama-nemotron-rerank-vl-1b-v2", "name": "llama-nemotron-rerank-vl-1b-v2", "description": "Reranking model for improving retrieval quality in search and recommendation systems", "family": "nemotron", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-03-31", "last_updated": "2026-03-31", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3.5-lightning-30b-a3b": {"id": "nvidia/nemotron-3.5-lightning-30b-a3b", "name": "Nemotron 3.5 Lightning 30B A3B", "description": "Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-08-11", "last_updated": "2026-08-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0, "output": 0}}, "nvidia/nv-embed-v1": {"id": "nvidia/nv-embed-v1", "name": "nv-embed-v1", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-06-07", "last_updated": "2025-07-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 2048}, "cost": {"input": 0, "output": 0}}, "nvidia/cosmos-predict1-5b": {"id": "nvidia/cosmos-predict1-5b", "name": "cosmos-predict1-5b", "description": "Video model for prompt-guided generation, editing, and motion workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-03-18", "last_updated": "2025-03-18", "modalities": {"input": ["text", "image", "video"], "output": ["video"]}, "open_weights": true, "limit": {"context": 0, "output": 4096}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-mini-4b-instruct": {"id": "nvidia/nemotron-mini-4b-instruct", "name": "nemotron-mini-4b-instruct", "description": "Compact Nemotron model for efficient reasoning and deployable AI agents", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-08-21", "last_updated": "2024-08-26", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "nvidia/usdcode": {"id": "nvidia/usdcode", "name": "usdcode", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-01-01", "last_updated": "2026-01-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "google/google-paligemma": {"id": "google/google-paligemma", "name": "paligemma", "description": "Gemini multimodal model for text, image, audio, video, and document tasks", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-05-14", "last_updated": "2024-08-26", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "google/gemma-3-4b-it": {"id": "google/gemma-3-4b-it", "name": "Gemma 3 4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-03-12", "last_updated": "2025-03-12", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0, "output": 0}}, "google/gemma-2-2b-it": {"id": "google/gemma-2-2b-it", "name": "Gemma 2 2b It", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-07-16", "last_updated": "2024-07-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "google/gemma-3-12b-it": {"id": "google/gemma-3-12b-it", "name": "Gemma 3 12B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-03-12", "last_updated": "2025-03-12", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0, "output": 0}}, "google/gemma-4-31b-it": {"id": "google/gemma-4-31b-it", "name": "Gemma-4-31B-IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 16384}, "cost": {"input": 0, "output": 0}}, "google/gemma-3n-e4b-it": {"id": "google/gemma-3n-e4b-it", "name": "Gemma 3n E4b It", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-06-03", "last_updated": "2025-06-03", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "google/gemma-3n-e2b-it": {"id": "google/gemma-3n-e2b-it", "name": "Gemma 3n E2b It", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06", "release_date": "2025-06-12", "last_updated": "2025-06-12", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "thinkingmachines/inkling": {"id": "thinkingmachines/inkling", "name": "Inkling", "description": "Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio", "family": "ling", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-07-15", "last_updated": "2026-07-15", "modalities": {"input": ["text", "image", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 16384}, "cost": {"input": 0, "output": 0}}, "baai/bge-m3": {"id": "baai/bge-m3", "name": "BGE M3", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "bge", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2024-01-30", "last_updated": "2026-04-30", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 8192, "output": 1024}, "cost": {"input": 0, "output": 0}}, "qwen/qwen3-coder-480b-a35b-instruct": {"id": "qwen/qwen3-coder-480b-a35b-instruct", "name": "Qwen3 Coder 480B A35B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 66536}, "cost": {"input": 0, "output": 0}}, "qwen/qwen-image-edit": {"id": "qwen/qwen-image-edit", "name": "Qwen Image Edit", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-08-19", "last_updated": "2025-08-19", "modalities": {"input": ["text", "image"], "output": ["image"]}, "open_weights": false, "limit": {"context": 0, "output": 0}, "cost": {"input": 0, "output": 0}}, "qwen/qwen2.5-coder-32b-instruct": {"id": "qwen/qwen2.5-coder-32b-instruct", "name": "Qwen2.5 Coder 32b Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-11-06", "last_updated": "2024-11-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "qwen/qwen-image": {"id": "qwen/qwen-image", "name": "Qwen Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": {"input": ["text", "image"], "output": ["image"]}, "open_weights": false, "limit": {"context": 0, "output": 0}, "cost": {"input": 0, "output": 0}}, "qwen/qwen3.5-397b-a17b": {"id": "qwen/qwen3.5-397b-a17b", "name": "Qwen3.5-397B-A17B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 8192}, "cost": {"input": 0, "output": 0}}, "qwen/qwen3.5-122b-a10b": {"id": "qwen/qwen3.5-122b-a10b", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image", "video", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0, "output": 0}}, "qwen/qwen3-next-80b-a3b-instruct": {"id": "qwen/qwen3-next-80b-a3b-instruct", "name": "Qwen3-Next-80B-A3B-Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2024-12-01", "last_updated": "2025-09-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 16384}, "cost": {"input": 0, "output": 0}}, "abacusai/dracarys-llama-3.1-70b-instruct": {"id": "abacusai/dracarys-llama-3.1-70b-instruct", "name": "dracarys-llama-3.1-70b-instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-09-11", "last_updated": "2025-05-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "upstage/solar-10.7b-instruct": {"id": "upstage/solar-10.7b-instruct", "name": "solar-10.7b-instruct", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-06-05", "last_updated": "2025-04-10", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "black-forest-labs/flux_2-klein-4b": {"id": "black-forest-labs/flux_2-klein-4b", "name": "FLUX.2 Klein 4B", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-06", "release_date": "2026-01-14", "last_updated": "2026-01-31", "modalities": {"input": ["image", "text"], "output": ["image"]}, "open_weights": true, "limit": {"context": 40960, "output": 40960}, "cost": {"input": 0, "output": 0}}, "black-forest-labs/flux_1-schnell": {"id": "black-forest-labs/flux_1-schnell", "name": "FLUX.1-schnell", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "knowledge": "2024-07", "release_date": "2024-08-01", "last_updated": "2026-02-04", "modalities": {"input": ["text"], "output": ["image"]}, "open_weights": true, "limit": {"context": 77, "input": 77, "output": 0}, "cost": {"input": 0, "output": 0}}, "black-forest-labs/flux_1-kontext-dev": {"id": "black-forest-labs/flux_1-kontext-dev", "name": "FLUX.1-Kontext-dev", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-08-12", "last_updated": "2025-08-12", "modalities": {"input": ["text", "image"], "output": ["image"]}, "open_weights": true, "limit": {"context": 40960, "output": 40960}, "cost": {"input": 0, "output": 0}}, "black-forest-labs/flux.1-dev": {"id": "black-forest-labs/flux.1-dev", "name": "FLUX.1-dev", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "flux", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-08", "release_date": "2024-08-01", "last_updated": "2025-09-05", "modalities": {"input": ["text"], "output": ["image"]}, "open_weights": false, "limit": {"context": 4096, "output": 0}, "cost": {"input": 0, "output": 0}}, "mistralai/mistral-medium-3.5-128b": {"id": "mistralai/mistral-medium-3.5-128b", "name": "Mistral Medium 3.5", "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0, "output": 0}}, "mistralai/mistral-nemotron": {"id": "mistralai/mistral-nemotron", "name": "mistral-nemotron", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-06-11", "last_updated": "2025-06-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "mistralai/mistral-medium-3-instruct": {"id": "mistralai/mistral-medium-3-instruct", "name": "Mistral Medium 3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "input": 131072, "output": 32768}, "cost": {"input": 0, "output": 0}}, "mistralai/mistral-small-4-119b-2603": {"id": "mistralai/mistral-small-4-119b-2603", "name": "mistral-small-4-119b-2603", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "mistralai/mistral-large-3-675b-instruct-2512": {"id": "mistralai/mistral-large-3-675b-instruct-2512", "name": "Mistral Large 3 675B Instruct 2512", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0, "output": 0}}, "mistralai/ministral-14b-instruct-2512": {"id": "mistralai/ministral-14b-instruct-2512", "name": "Ministral 3 14B Instruct 2512", "description": "Compact Mistral VLM for chat and instruction-based workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 16384}, "cost": {"input": 0, "output": 0}}, "mistralai/mixtral-8x22b-instruct": {"id": "mistralai/mixtral-8x22b-instruct", "name": "Mistral: Mixtral 8x22B Instruct", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2024-04-17", "last_updated": "2024-04-17", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 65536, "output": 13108}, "cost": {"input": 0, "output": 0}}, "mistralai/magistral-small-2506": {"id": "mistralai/magistral-small-2506", "name": "Magistral Small 2506", "description": "Mistral reasoning model for transparent analysis, math, and complex decisions", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "release_date": "2025-09-25", "last_updated": "2025-09-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 32768, "input": 32768, "output": 32768}, "cost": {"input": 0, "output": 0}}, "mistralai/mixtral-8x7b-instruct": {"id": "mistralai/mixtral-8x7b-instruct", "name": "Mistral: Mixtral 8x7B Instruct", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2023-12-10", "last_updated": "2026-03-15", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 16384}, "cost": {"input": 0, "output": 0}}, "mistralai/mistral-7b-instruct-v0.3": {"id": "mistralai/mistral-7b-instruct-v0.3", "name": "Mistral-7B-Instruct-v0.3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-04-01", "last_updated": "2025-04-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 65536, "output": 65536}, "cost": {"input": 0, "output": 0}}, "bytedance/seed-oss-36b-instruct": {"id": "bytedance/seed-oss-36b-instruct", "name": "ByteDance-Seed/Seed-OSS-36B-Instruct", "description": "Tool-capable chat model for instruction following and agentic application workflows", "family": "seed", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-09-04", "last_updated": "2025-11-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262000, "output": 262000}, "cost": {"input": 0, "output": 0}}, "meta/llama-3.1-70b-instruct": {"id": "meta/llama-3.1-70b-instruct", "name": "Llama 3.1 70b Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-07-16", "last_updated": "2024-07-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "meta/llama-4-maverick-17b-128e-instruct": {"id": "meta/llama-4-maverick-17b-128e-instruct", "name": "Llama 4 Maverick 17b 128e Instruct", "description": "Open multimodal Llama model for strong reasoning and fast responses", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-02", "release_date": "2025-04-01", "last_updated": "2025-04-01", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "meta/llama-guard-4-12b": {"id": "meta/llama-guard-4-12b", "name": "Llama Guard 4 12B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-04-05", "last_updated": "2026-04-30", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 0, "output": 0}}, "meta/llama-3.2-1b-instruct": {"id": "meta/llama-3.2-1b-instruct", "name": "Llama 3.2 1b Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-18", "last_updated": "2024-09-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "meta/llama-3.3-70b-instruct": {"id": "meta/llama-3.3-70b-instruct", "name": "Llama 3.3 70b Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-11-26", "last_updated": "2024-11-26", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "meta/llama-3.2-3b-instruct": {"id": "meta/llama-3.2-3b-instruct", "name": "Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2024-09-18", "last_updated": "2024-09-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 32000}, "cost": {"input": 0, "output": 0}}, "meta/llama-3.2-90b-vision-instruct": {"id": "meta/llama-3.2-90b-vision-instruct", "name": "Llama-3.2-90B-Vision-Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "meta/esmfold": {"id": "meta/esmfold", "name": "esmfold", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-03-15", "last_updated": "2025-06-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "meta/llama-3.1-8b-instruct": {"id": "meta/llama-3.1-8b-instruct", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 16000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "meta/llama-3.2-11b-vision-instruct": {"id": "meta/llama-3.2-11b-vision-instruct", "name": "Llama 3.2 11b Vision Instruct", "description": "Open Llama multimodal model for image understanding and text reasoning", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-09-18", "last_updated": "2024-09-18", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 0, "output": 0}}, "meta/esm2-650m": {"id": "meta/esm2-650m", "name": "esm2-650m", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-08-29", "last_updated": "2025-03-10", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "poolside/laguna-xs-2.1": {"id": "poolside/laguna-xs-2.1", "name": "Laguna XS 2.1", "description": "Agentic coding model from Poolside in the XS size class for local deployment", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-02", "last_updated": "2026-07-02", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 16384}, "cost": {"input": 0, "output": 0}}, "deepseek-ai/deepseek-v4-flash": {"id": "deepseek-ai/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 393216}, "cost": {"input": 0.14, "output": 0.28, "cache_read": 0.0028}}, "deepseek-ai/deepseek-v4-pro": {"id": "deepseek-ai/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 393216}, "cost": {"input": 0.435, "output": 0.87, "cache_read": 0.003625}}, "stepfun-ai/step-3.5-flash": {"id": "stepfun-ai/step-3.5-flash", "name": "Step 3.5 Flash", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "temperature": true, "release_date": "2026-02-02", "last_updated": "2026-02-02", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 16384}, "cost": {"input": 0, "output": 0}}, "stepfun-ai/step-3.7-flash": {"id": "stepfun-ai/step-3.7-flash", "name": "Step 3.7 Flash", "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "temperature": true, "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 16384}, "cost": {"input": 0, "output": 0}}, "z-ai/glm-5.2": {"id": "z-ai/glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 131072}, "cost": {"input": 0, "output": 0}}, "moonshotai/kimi-k2.6": {"id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "status": "deprecated", "cost": {"input": 0, "output": 0}}, "moonshotai/kimi-k2-instruct-0905": {"id": "moonshotai/kimi-k2-instruct-0905", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "status": "deprecated", "cost": {"input": 0, "output": 0}}, "openai/gpt-oss-20b": {"id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0, "output": 0}}, "openai/whisper-large-v3": {"id": "openai/whisper-large-v3", "name": "Whisper Large v3", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2023-09", "release_date": "2023-09-01", "last_updated": "2025-09-05", "modalities": {"input": ["audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 0, "output": 4096}, "cost": {"input": 0, "output": 0}}, "openai/gpt-oss-120b": {"id": "openai/gpt-oss-120b", "name": "GPT-OSS-120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08", "release_date": "2025-08-04", "last_updated": "2025-08-14", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "minimaxai/minimax-m2.7": {"id": "minimaxai/minimax-m2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-04-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0, "output": 0}}, "minimaxai/minimax-m3": {"id": "minimaxai/minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 16384}, "cost": {"input": 0, "output": 0}}, "sarvamai/sarvam-m": {"id": "sarvamai/sarvam-m", "name": "sarvam-m", "description": "Efficient Indian-language reasoning model for chat, coding, and multilingual work", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}}}, "google": {"id": "google", "env": ["GOOGLE_API_KEY", "GOOGLE_GENERATIVE_AI_API_KEY", "GEMINI_API_KEY"], "npm": "@ai-sdk/google", "name": "Google", "doc": "https://ai.google.dev/gemini-api/docs/models", "models": {"gemini-2.5-computer-use-preview-10-2025": {"id": "gemini-2.5-computer-use-preview-10-2025", "name": "Gemini 2.5 Computer Use Preview 10-2025", "description": "Specialized Gemini 2.5 model for browser-control agents that automate UI tasks", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-10-07", "last_updated": "2025-10-07", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 1.25, "output": 10, "tiers": [{"input": 2.5, "output": 15, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 15}}}, "deep-research-preview-04-2026": {"id": "deep-research-preview-04-2026", "name": "Deep Research Preview (Apr-21-2026)", "description": "Agentic model for autonomous multi-step research, synthesis, and cited reports", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text", "image"]}, "open_weights": false, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 2, "output": 12, "cache_read": 0.2, "tiers": [{"input": 4, "output": 18, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 18, "cache_read": 0.4}}}, "gemini-3.1-flash-tts-preview": {"id": "gemini-3.1-flash-tts-preview", "name": "Gemini 3.1 Flash TTS Preview", "description": "Low-latency speech generation with steerable prompts and expressive audio tags", "family": "gemini-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-15", "last_updated": "2026-04-15", "modalities": {"input": ["text"], "output": ["audio"]}, "open_weights": false, "limit": {"context": 8192, "output": 16384}, "cost": {"input": 1, "output": 20}}, "gemini-flash-latest": {"id": "gemini-flash-latest", "name": "Gemini Flash Latest", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 1.5, "output": 9, "cache_read": 0.15, "input_audio": 1.5}}, "gemini-embedding-2": {"id": "gemini-embedding-2", "name": "Gemini Embedding 2", "description": "Multimodal embedding model mapping text, images, video, audio, and PDFs into a unified embedding space", "family": "gemini", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2025-11", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": {"input": ["text", "image", "audio", "video", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 8192, "output": 1}, "cost": {"input": 0.2, "output": 0, "input_audio": 6.5}}, "lyria-3-pro-preview": {"id": "lyria-3-pro-preview", "name": "Lyria 3 Pro Preview", "description": "Music generation model for full-length songs from text or images with vocals and structure", "family": "lyria", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-03-25", "last_updated": "2026-03-25", "modalities": {"input": ["text", "image"], "output": ["text", "audio"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0, "output": 0}}, "gemini-3.5-flash": {"id": "gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 1.5, "output": 9, "cache_read": 0.15, "input_audio": 1.5}}, "gemini-2.5-flash": {"id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 0, "max": 24576}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": {"input": ["text", "image", "audio", "video", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0.3, "output": 2.5, "cache_read": 0.03, "input_audio": 1}}, "gemini-3.5-flash-lite": {"id": "gemini-3.5-flash-lite", "name": "Gemini 3.5 Flash Lite", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-03", "release_date": "2026-07-21", "last_updated": "2026-07-21", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0.3, "output": 2.5, "cache_read": 0.03}}, "lyria-3-clip-preview": {"id": "lyria-3-clip-preview", "name": "Lyria 3 Clip Preview", "description": "Music generation model for short 30-second clips, loops, and previews from text or image prompts", "family": "lyria", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-03-25", "last_updated": "2026-03-25", "modalities": {"input": ["text", "image"], "output": ["text", "audio"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0, "output": 0}}, "gemini-omni-flash-preview": {"id": "gemini-omni-flash-preview", "name": "Gemini Omni Flash Preview", "description": "Video generation and editing model for fast, conversational text- and image-to-video workflows", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": {"input": ["text", "image", "video"], "output": ["video"]}, "open_weights": false, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 1.5, "output": 17.5}}, "veo-3.1-generate-preview": {"id": "veo-3.1-generate-preview", "name": "Veo 3.1", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-10-15", "last_updated": "2026-01", "modalities": {"input": ["text", "image"], "output": ["video"]}, "open_weights": false, "limit": {"context": 480, "output": 8192}, "status": "beta"}, "deep-research-max-preview-04-2026": {"id": "deep-research-max-preview-04-2026", "name": "Deep Research Max Preview (Apr-21-2026)", "description": "Maximum-comprehensiveness agentic researcher for multi-step investigation, synthesis, and cited reports", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text", "image"]}, "open_weights": false, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 2, "output": 12, "cache_read": 0.2, "tiers": [{"input": 4, "output": 18, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 18, "cache_read": 0.4}}}, "gemini-3-pro-image-preview": {"id": "gemini-3-pro-image-preview", "name": "Nano Banana Pro", "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-11-20", "last_updated": "2025-11-20", "modalities": {"input": ["text", "image"], "output": ["text", "image"]}, "open_weights": false, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 2, "output": 120}}, "gemini-3.1-flash-lite-preview": {"id": "gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", "description": "Legacy model retained for compatibility with older integrations", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "status": "deprecated", "cost": {"input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5}}, "gemini-3.5-live-translate-preview": {"id": "gemini-3.5-live-translate-preview", "name": "Gemini 3.5 Live Translate Preview", "description": "Low-latency audio-to-audio model for real-time speech translation across 70+ languages", "family": "gemini-pro", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": {"input": ["audio"], "output": ["audio", "text"]}, "open_weights": false, "limit": {"context": 16384, "output": 32768}, "cost": {"input": 3.5, "output": 21, "input_audio": 3.5, "output_audio": 21}}, "gemini-3-flash-preview": {"id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0.5, "output": 3, "cache_read": 0.05, "input_audio": 1}}, "gemini-3.1-pro-preview-customtools": {"id": "gemini-3.1-pro-preview-customtools", "name": "Gemini 3.1 Pro Preview Custom Tools", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 2, "output": 12, "cache_read": 0.2, "tiers": [{"input": 4, "output": 18, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 18, "cache_read": 0.4}}}, "gemini-3.1-flash-lite-image": {"id": "gemini-3.1-flash-lite-image", "name": "Nano Banana 2 Lite", "description": "Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "high"]}], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": {"input": ["text", "image"], "output": ["text", "image"]}, "open_weights": false, "limit": {"context": 65536, "output": 65536}, "cost": {"input": 0.25, "output": 30}}, "gemini-3.1-flash-image-preview": {"name": "Nano Banana 2 (Gemini 3.1 Flash Image)", "release_date": "2026-02-26", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0.25, "output": 1.5}, "id": "gemini-3.1-flash-image-preview"}, "gemini-robotics-er-1.6-preview": {"id": "gemini-robotics-er-1.6-preview", "name": "Gemini Robotics-ER 1.6 Preview", "description": "Vision-language model for embodied reasoning: spatial understanding, task planning, and physical-world agentic robotics", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 0}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-14", "last_updated": "2026-04-14", "modalities": {"input": ["text", "image", "video", "audio"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 1, "output": 5, "input_audio": 2}}, "gemma-4-26b-a4b-it": {"id": "gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}}, "gemini-embedding-001": {"id": "gemini-embedding-001", "name": "Gemini Embedding 001", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "gemini", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2025-05", "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 2048, "output": 1}, "cost": {"input": 0.15, "output": 0}}, "veo-3.1-lite-generate-preview": {"id": "veo-3.1-lite-generate-preview", "name": "Veo 3.1 lite", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-03-31", "last_updated": "2026-03-31", "modalities": {"input": ["text", "image"], "output": ["video"]}, "open_weights": false, "limit": {"context": 480, "output": 8192}}, "gemini-3.6-flash": {"id": "gemini-3.6-flash", "name": "Gemini 3.6 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-03", "release_date": "2026-07-21", "last_updated": "2026-07-21", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 1.5, "output": 7.5, "cache_read": 0.15, "input_audio": 1.5}}, "veo-3.1-fast-generate-preview": {"id": "veo-3.1-fast-generate-preview", "name": "Veo 3.1 fast", "description": "Video model for prompt-guided generation, editing, and motion workflows", "family": "veo", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-10-15", "last_updated": "2026-01-01", "modalities": {"input": ["text", "image", "video"], "output": ["video"]}, "open_weights": false, "limit": {"context": 480, "output": 8192}}, "gemini-3.1-flash-lite": {"id": "gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5}}, "gemma-4-31b-it": {"id": "gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}}, "gemini-3.1-flash-image": {"id": "gemini-3.1-flash-image", "name": "Nano Banana 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "high"]}], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": {"input": ["text", "image", "video", "pdf"], "output": ["text", "image"]}, "open_weights": false, "limit": {"context": 65536, "output": 65536}, "cost": {"input": 0.5, "output": 60}}, "gemini-2.5-flash-image": {"id": "gemini-2.5-flash-image", "name": "Nano Banana", "description": "Nano Banana image model for fast generation, edits, and character-consistent assets", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "temperature": true, "knowledge": "2024-06", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": {"input": ["text", "image"], "output": ["text", "image"]}, "open_weights": false, "limit": {"context": 32768, "output": 32768}, "cost": {"input": 0.3, "output": 30, "cache_read": 0.075}}, "gemini-2.5-flash-preview-tts": {"id": "gemini-2.5-flash-preview-tts", "name": "Gemini 2.5 Flash Preview TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gemini-flash", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-05-01", "last_updated": "2025-05-01", "modalities": {"input": ["text"], "output": ["audio"]}, "open_weights": false, "limit": {"context": 8192, "output": 16384}, "cost": {"input": 0.5, "output": 10}}, "gemini-3-pro-image": {"id": "gemini-3-pro-image", "name": "Nano Banana Pro", "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "high"]}], "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": {"input": ["text", "image"], "output": ["text", "image"]}, "open_weights": false, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 2, "output": 120}}, "gemini-3.1-pro-preview": {"id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 2, "output": 12, "cache_read": 0.2, "tiers": [{"input": 4, "output": 18, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 18, "cache_read": 0.4}}}, "gemini-flash-lite-latest": {"id": "gemini-flash-lite-latest", "name": "Gemini Flash-Lite Latest", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0.25, "output": 1.5, "cache_read": 0.025, "input_audio": 0.5}}, "gemini-2.5-pro": {"id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 128, "max": 32768}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": {"input": ["text", "image", "audio", "video", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 1.25, "output": 10, "cache_read": 0.125, "tiers": [{"input": 2.5, "output": 15, "cache_read": 0.25, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 15, "cache_read": 0.25}}}, "gemini-2.5-flash-lite": {"id": "gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 512, "max": 24576}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": {"input": ["text", "image", "audio", "video", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0.1, "output": 0.4, "cache_read": 0.01, "input_audio": 0.3}}, "gemini-2.5-pro-preview-tts": {"id": "gemini-2.5-pro-preview-tts", "name": "Gemini 2.5 Pro Preview TTS", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gemini-flash", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2025-01", "release_date": "2025-05-01", "last_updated": "2025-05-01", "modalities": {"input": ["text"], "output": ["audio"]}, "open_weights": false, "limit": {"context": 8192, "output": 16384}, "cost": {"input": 1, "output": 20}}, "gemini-3.1-flash-live-preview": {"id": "gemini-3.1-flash-live-preview", "name": "Gemini 3.1 Flash Live Preview", "description": "High-quality, low-latency Live API model for real-time dialogue and voice-first AI applications", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-26", "last_updated": "2026-03-26", "modalities": {"input": ["text", "image", "video", "audio"], "output": ["text", "audio"]}, "open_weights": false, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 0.75, "output": 4.5, "input_audio": 3, "output_audio": 12}}}}, "github-copilot": {"id": "github-copilot", "env": ["GITHUB_TOKEN"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.githubcopilot.com", "name": "GitHub Copilot", "doc": "https://docs.github.com/en/copilot", "models": {"gemini-3.5-flash": {"id": "gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}, {"type": "budget_tokens", "min": 256, "max": 24000}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "input": 128000, "output": 64000}, "cost": {"input": 1.5, "output": 9, "cache_read": 0.15, "input_audio": 1.5}}, "claude-sonnet-4.6": {"id": "claude-sonnet-4.6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "max"]}, {"type": "budget_tokens", "min": 1024, "max": 32000}], "tool_call": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "input": 168000, "output": 32000}, "cost": {"input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75}}, "gpt-5.6-sol": {"id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", "family": "gpt-sol", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25, "tiers": [{"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5}}}, "claude-fable-5": {"id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5}}, "mai-code-1-flash-picker": {"id": "mai-code-1-flash-picker", "name": "MAI-Code-1-Flash", "description": "Microsoft coding model built for fast, efficient assistance in everyday developer workflows", "family": "mai", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-06-02", "last_updated": "2026-06-08", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 256000, "input": 128000, "output": 128000}, "cost": {"input": 0.75, "output": 4.5, "cache_read": 0.075}}, "claude-opus-4.5": {"id": "claude-opus-4.5", "name": "Claude Opus 4.5 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 1024, "max": 32000}], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "input": 168000, "output": 32000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "gpt-5.5": {"id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 5, "output": 30, "cache_read": 0.5, "tiers": [{"input": 10, "output": 45, "cache_read": 1, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 10, "output": 45, "cache_read": 1}}}, "mai-code-1.1-flash": {"id": "mai-code-1.1-flash", "name": "MAI-Code-1.1-Flash", "description": "Microsoft coding model with native vision support, optimized for fast and efficient software development", "family": "mai", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "release_date": "2026-08-11", "last_updated": "2026-08-11", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 256000, "input": 128000, "output": 128000}, "cost": {"input": 0.2, "output": 1.2, "cache_read": 0.02}}, "claude-opus-4.7": {"id": "claude-opus-4.7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "input": 168000, "output": 32000}, "experimental": {"modes": {"fast": {"cost": {"input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5}, "provider": {"body": {"speed": "fast"}, "headers": {"anthropic-beta": "fast-mode-2026-02-01"}}}}}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "claude-sonnet-4.5": {"id": "claude-sonnet-4.5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 1024, "max": 32000}], "tool_call": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "input": 168000, "output": 32000}, "cost": {"input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75}}, "gpt-5.4": {"id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [{"input": 5, "output": 22.5, "cache_read": 0.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 5, "output": 22.5, "cache_read": 0.5}}}, "gpt-5.2-codex": {"id": "gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "gpt-5.4-nano": {"id": "gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 0.2, "output": 1.25, "cache_read": 0.02}}, "gemini-3.6-flash": {"id": "gemini-3.6-flash", "name": "Gemini 3.6 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}, {"type": "budget_tokens", "min": 256, "max": 32000}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-03", "release_date": "2026-07-21", "last_updated": "2026-07-21", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "input": 936000, "output": 64000}, "cost": {"input": 1.5, "output": 7.5, "cache_read": 0.15}}, "claude-sonnet-4": {"id": "claude-sonnet-4", "name": "Claude Sonnet 4 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 216000, "input": 128000, "output": 16000}, "cost": {"input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75}}, "gpt-5.4-mini": {"id": "gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 0.75, "output": 4.5, "cache_read": 0.075}}, "gpt-5.6-luna": {"id": "gpt-5.6-luna", "name": "GPT-5.6 Luna", "description": "Cost-efficient GPT-5.6 model for fast, high-volume workloads", "family": "gpt-luna", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 0.2, "output": 1.2, "cache_read": 0.02, "tiers": [{"input": 0.4, "output": 1.8, "cache_read": 0.04, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 0.4, "output": 1.8, "cache_read": 0.04}}}, "gpt-5.2": {"id": "gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "claude-haiku-4.5": {"id": "claude-haiku-4.5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 1024, "max": 32000}], "tool_call": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "input": 136000, "output": 64000}, "cost": {"input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25}}, "gpt-5.3-codex": {"id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "gpt-5-mini": {"id": "gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 264000, "input": 128000, "output": 64000}, "cost": {"input": 0.25, "output": 2, "cache_read": 0.025}}, "grok-4.5": {"id": "grok-4.5", "name": "Grok 4.5", "description": "xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 500000, "input": 372000, "output": 128000}, "cost": {"input": 2, "output": 6, "cache_read": 0.5, "tiers": [{"input": 4, "output": 12, "cache_read": 1, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 12, "cache_read": 1}}}, "kimi-k2.7-code": {"id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "input": 224000, "output": 32000}, "cost": {"input": 0.95, "output": 4, "cache_read": 0.19}}, "gemini-3.1-pro-preview": {"id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}, {"type": "budget_tokens", "min": 256, "max": 32000}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "input": 936000, "output": 64000}, "cost": {"input": 2, "output": 12, "cache_read": 0.2, "tiers": [{"input": 4, "output": 18, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 18, "cache_read": 0.4}}}, "kimi-k3": {"id": "kimi-k3", "name": "Kimi K3", "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work", "family": "kimi-k3", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "high", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-07-16", "last_updated": "2026-07-16", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 131072}, "cost": {"input": 3, "output": 15, "cache_read": 0.3}}, "claude-opus-4.8": {"id": "claude-opus-4.8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "input": 168000, "output": 64000}, "experimental": {"modes": {"fast": {"cost": {"input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5}, "provider": {"body": {"speed": "fast"}, "headers": {"anthropic-beta": "fast-mode-2026-02-01"}}}}}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "gpt-5.6-terra": {"id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", "description": "Balanced GPT-5.6 model for capable, cost-efficient everyday work", "family": "gpt-terra", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 2, "output": 12, "cache_read": 0.2, "tiers": [{"input": 4, "output": 18, "cache_read": 0.4, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 4, "output": 18, "cache_read": 0.4}}}, "claude-sonnet-5": {"id": "claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5}}, "gpt-4.1": {"id": "gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "input": 128000, "output": 16384}, "cost": {"input": 2, "output": 8, "cache_read": 0.5}}, "claude-opus-5": {"id": "claude-opus-5", "name": "Claude Opus 5", "description": "Strongest Claude Opus model for coding, agents, and professional work", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-05", "release_date": "2026-07-24", "last_updated": "2026-07-24", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "input": 936000, "output": 64000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "claude-opus-4.6": {"id": "claude-opus-4.6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "max"]}], "tool_call": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "input": 168000, "output": 32000}, "experimental": {"modes": {"fast": {"cost": {"input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5}, "provider": {"body": {"speed": "fast"}, "headers": {"anthropic-beta": "fast-mode-2026-02-01"}}}}}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}}}, "lmstudio": {"id": "lmstudio", "env": ["LMSTUDIO_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "http://127.0.0.1:1234/v1", "name": "LMStudio", "doc": "https://lmstudio.ai/models", "models": {"qwen/qwen3-coder-30b": {"id": "qwen/qwen3-coder-30b", "name": "Qwen3 Coder 30B", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0, "output": 0}}, "qwen/qwen3-30b-a3b-2507": {"id": "qwen/qwen3-30b-a3b-2507", "name": "Qwen3 30B A3B 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-30", "last_updated": "2025-07-30", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 16384}, "cost": {"input": 0, "output": 0}}, "openai/gpt-oss-20b": {"id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0, "output": 0}}}}, "xai": {"id": "xai", "env": ["XAI_API_KEY"], "npm": "@ai-sdk/xai", "name": "xAI", "doc": "https://docs.x.ai/docs/models", "models": {"grok-imagine-video": {"id": "grok-imagine-video", "name": "Grok Imagine Video", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "grok", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-01-28", "last_updated": "2026-01-28", "modalities": {"input": ["text", "image", "video", "pdf"], "output": ["video"]}, "open_weights": false, "limit": {"context": 1024, "output": 0}}, "grok-4.3": {"id": "grok-4.3", "name": "Grok 4.3", "description": "xAI's Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 30000}, "cost": {"input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [{"input": 2.5, "output": 5, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 5, "cache_read": 0.4}}}, "grok-4.20-0309-non-reasoning": {"id": "grok-4.20-0309-non-reasoning", "name": "Grok 4.20 (Non-Reasoning)", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-09", "last_updated": "2026-03-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 30000}, "cost": {"input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [{"input": 2.5, "output": 5, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 5, "cache_read": 0.4}}}, "grok-imagine-video-1.5": {"id": "grok-imagine-video-1.5", "name": "Grok Imagine Video 1.5", "description": "Video model for image-to-video generation, editing, and extension workflows", "family": "grok", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-05-30", "last_updated": "2026-05-30", "modalities": {"input": ["text", "image", "audio", "pdf"], "output": ["video"]}, "open_weights": false, "limit": {"context": 1024, "output": 0}}, "grok-4.20-multi-agent-0309": {"id": "grok-4.20-multi-agent-0309", "name": "Grok 4.20 Multi-Agent", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2026-03-09", "last_updated": "2026-03-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 30000}, "cost": {"input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [{"input": 2.5, "output": 5, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 5, "cache_read": 0.4}}}, "grok-imagine-image-quality": {"id": "grok-imagine-image-quality", "name": "Grok Imagine Image Quality", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "grok", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-04-03", "last_updated": "2026-04-03", "modalities": {"input": ["text", "image", "pdf"], "output": ["image", "pdf"]}, "open_weights": false, "limit": {"context": 8000, "output": 0}}, "grok-4.5": {"id": "grok-4.5", "name": "Grok 4.5", "description": "xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 500000, "output": 500000}, "cost": {"input": 2, "output": 6, "cache_read": 0.3, "tiers": [{"input": 4, "output": 12, "cache_read": 0.6, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 12, "cache_read": 0.6}}}, "grok-4.6": {"id": "grok-4.6", "name": "Grok 4.6", "description": "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-02-01", "release_date": "2026-08-12", "last_updated": "2026-08-12", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 500000, "output": 500000}, "experimental": {"modes": {"fast": {"cost": {"input": 4, "output": 12, "cache_read": 1}, "provider": {"body": {"service_tier": "priority"}}}}}, "cost": {"input": 2, "output": 6, "cache_read": 0.5, "tiers": [{"input": 4, "output": 12, "cache_read": 1, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 12, "cache_read": 1}}}, "grok-build-0.1": {"id": "grok-build-0.1", "name": "Grok Build 0.1", "description": "Fast Grok coding model tuned for agentic engineering and iterative edits", "family": "grok-build", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 256000, "output": 256000}, "cost": {"input": 1, "output": 2, "cache_read": 0.2, "tiers": [{"input": 2, "output": 4, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2, "output": 4, "cache_read": 0.4}}}, "grok-4.20-0309-reasoning": {"id": "grok-4.20-0309-reasoning", "name": "Grok 4.20 (Reasoning)", "description": "Reasoning Grok for document-heavy analysis and long-horizon tool use", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-09", "last_updated": "2026-03-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 30000}, "cost": {"input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [{"input": 2.5, "output": 5, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 5, "cache_read": 0.4}}}, "grok-imagine-image": {"id": "grok-imagine-image", "name": "Grok Imagine Image", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "grok", "attachment": true, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-01-28", "last_updated": "2026-01-28", "modalities": {"input": ["text", "image", "pdf"], "output": ["image", "pdf"]}, "open_weights": false, "limit": {"context": 8000, "output": 0}}}}, "mistral": {"id": "mistral", "env": ["MISTRAL_API_KEY"], "npm": "@ai-sdk/mistral", "name": "Mistral", "doc": "https://docs.mistral.ai/getting-started/models/", "models": {"mistral-small-2506": {"id": "mistral-small-2506", "name": "Mistral Small 3.2", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-03", "release_date": "2025-06-20", "last_updated": "2025-06-20", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 0.1, "output": 0.3}}, "pixtral-large-latest": {"id": "pixtral-large-latest", "name": "Pixtral Large (latest)", "description": "Mistral's larger vision model for document-heavy image understanding and chat", "family": "pixtral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2024-11-04", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 2, "output": 6}}, "pixtral-12b": {"id": "pixtral-12b", "name": "Pixtral 12B", "description": "Mistral vision-language model for image understanding and multimodal chat", "family": "pixtral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-09", "release_date": "2024-09-01", "last_updated": "2024-09-01", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 0.15, "output": 0.15}}, "open-mixtral-8x7b": {"id": "open-mixtral-8x7b", "name": "Mixtral 8x7B", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mixtral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-01", "release_date": "2023-12-11", "last_updated": "2023-12-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32000, "output": 32000}, "cost": {"input": 0.7, "output": 0.7}}, "labs-devstral-small-2512": {"id": "labs-devstral-small-2512", "name": "Devstral Small 2", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 256000}, "status": "deprecated", "cost": {"input": 0, "output": 0}}, "magistral-medium-latest": {"id": "magistral-medium-latest", "name": "Magistral Medium (latest)", "description": "Mistral reasoning model for transparent analysis, math, and complex decisions", "family": "magistral-medium", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-03-17", "last_updated": "2025-03-20", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 2, "output": 5}}, "mistral-medium-2508": {"id": "mistral-medium-2508", "name": "Mistral Medium 3.1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-08-12", "last_updated": "2025-08-12", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.4, "output": 2}}, "mistral-large-2411": {"id": "mistral-large-2411", "name": "Mistral Large 2.1", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-18", "last_updated": "2024-11-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 2, "output": 6}}, "voxtral-mini-tts-latest": {"id": "voxtral-mini-tts-latest", "name": "Voxtral Mini TTS (latest)", "description": "Multilingual text-to-speech model with zero-shot voice cloning", "family": "voxtral", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-03-01", "last_updated": "2026-03-01", "modalities": {"input": ["text"], "output": ["audio"]}, "open_weights": false, "limit": {"context": 0, "output": 0}}, "mistral-medium-latest": {"id": "mistral-medium-latest", "name": "Mistral Medium (latest)", "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 1.5, "output": 7.5}}, "devstral-small-2507": {"id": "devstral-small-2507", "name": "Devstral Small", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-07-10", "last_updated": "2025-07-10", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "status": "deprecated", "cost": {"input": 0.1, "output": 0.3}}, "open-mistral-7b": {"id": "open-mistral-7b", "name": "Mistral 7B", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2023-09-27", "last_updated": "2023-09-27", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 8000, "output": 8000}, "cost": {"input": 0.25, "output": 0.25}}, "mistral-medium-2505": {"id": "mistral-medium-2505", "name": "Mistral Medium 3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.4, "output": 2}}, "ministral-3b-latest": {"id": "ministral-3b-latest", "name": "Ministral 3B (latest)", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-10-01", "last_updated": "2024-10-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 0.04, "output": 0.04}}, "mistral-small-latest": {"id": "mistral-small-latest", "name": "Mistral Small (latest)", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "high"]}], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 256000}, "cost": {"input": 0.15, "output": 0.6}}, "open-mixtral-8x22b": {"id": "open-mixtral-8x22b", "name": "Mixtral 8x22B", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mixtral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-04-17", "last_updated": "2024-04-17", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 64000, "output": 64000}, "cost": {"input": 2, "output": 6}}, "mistral-nemo": {"id": "mistral-nemo", "name": "Mistral Nemo", "description": "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 0.15, "output": 0.15}}, "mistral-small-2603": {"id": "mistral-small-2603", "name": "Mistral Small 4", "description": "Fast Mistral production model for chat, extraction, and cost-sensitive agents", "family": "mistral-small", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "high"]}], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 256000}, "cost": {"input": 0.15, "output": 0.6}}, "mistral-medium-2604": {"id": "mistral-medium-2604", "name": "Mistral Medium 3.5", "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-29", "last_updated": "2026-04-29", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 1.5, "output": 7.5}}, "voxtral-mini-latest": {"id": "voxtral-mini-latest", "name": "Voxtral Mini Chat", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": {"input": ["audio", "text"], "output": ["text"]}, "open_weights": false, "cost": {"input": 0.04, "output": 0.04}, "limit": {"context": 16384, "output": 16384}}, "voxtral-small-latest": {"id": "voxtral-small-latest", "name": "Voxtral Small (latest)", "description": "Instruct model with native audio input for speech understanding and tool use", "family": "voxtral", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "release_date": "2025-07-15", "last_updated": "2025-07-15", "modalities": {"input": ["text", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32000, "output": 32000}, "cost": {"input": 0.1, "output": 0.3}}, "devstral-latest": {"id": "devstral-latest", "name": "Devstral 2", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "status": "deprecated", "cost": {"input": 0.4, "output": 2}}, "ministral-8b-latest": {"id": "ministral-8b-latest", "name": "Ministral 8B (latest)", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-10-01", "last_updated": "2024-10-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 0.1, "output": 0.1}}, "mistral-embed": {"id": "mistral-embed", "name": "Mistral Embed", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "mistral-embed", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2023-12-11", "last_updated": "2023-12-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 8000, "output": 3072}, "cost": {"input": 0.1, "output": 0}}, "codestral-latest": {"id": "codestral-latest", "name": "Codestral (latest)", "description": "Mistral code model for completions, refactors, and developer IDE workflows", "family": "codestral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2024-05-29", "last_updated": "2025-01-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 4096}, "cost": {"input": 0.3, "output": 0.9}}, "devstral-medium-latest": {"id": "devstral-medium-latest", "name": "Devstral 2 (latest)", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "status": "deprecated", "cost": {"input": 0.4, "output": 2}}, "devstral-2512": {"id": "devstral-2512", "name": "Devstral 2", "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-12", "release_date": "2025-12-09", "last_updated": "2025-12-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "status": "deprecated", "cost": {"input": 0.4, "output": 2}}, "mistral-large-latest": {"id": "mistral-large-latest", "name": "Mistral Large (latest)", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.5, "output": 1.5}}, "devstral-medium-2507": {"id": "devstral-medium-2507", "name": "Devstral Medium", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-07-10", "last_updated": "2025-07-10", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "status": "deprecated", "cost": {"input": 0.4, "output": 2}}, "devstral-small-2505": {"id": "devstral-small-2505", "name": "Devstral Small 2505", "description": "Legacy model retained for compatibility with older integrations", "family": "devstral", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "status": "deprecated", "cost": {"input": 0.1, "output": 0.3}}, "magistral-small": {"id": "magistral-small", "name": "Magistral Small", "description": "Mistral reasoning model for transparent analysis, math, and complex decisions", "family": "magistral-small", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-06", "release_date": "2025-03-17", "last_updated": "2025-03-17", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 0.5, "output": 1.5}}, "open-mistral-nemo": {"id": "open-mistral-nemo", "name": "Open Mistral Nemo", "description": "Legacy model retained for compatibility with older integrations", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "status": "deprecated", "cost": {"input": 0.15, "output": 0.15}}, "mistral-large-2512": {"id": "mistral-large-2512", "name": "Mistral Large 3", "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.5, "output": 1.5}}, "voxtral-mini-transcription": {"id": "voxtral-mini-transcription", "name": "Voxtral Mini Transcription", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": {"input": ["audio"], "output": ["text"]}, "open_weights": false, "cost": {"input": 0.003, "output": 0}, "limit": {"context": 16384, "output": 16384}}}}, "zai-coding-plan": {"id": "zai-coding-plan", "env": ["ZHIPU_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.z.ai/api/coding/paas/v4", "name": "Z.AI Coding Plan", "doc": "https://docs.z.ai/devpack/overview", "models": {"glm-5.2": {"id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 131072}, "cost": {"input": 0, "output": 0, "cache_read": 0, "cache_write": 0}}, "glm-5.2-highspeed": {"id": "glm-5.2-highspeed", "name": "GLM-5.2 Highspeed", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 131072}, "cost": {"input": 0, "output": 0, "cache_read": 0, "cache_write": 0}}, "glm-4.7": {"id": "glm-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0, "output": 0, "cache_read": 0, "cache_write": 0}}, "glm-5-turbo": {"id": "glm-5-turbo", "name": "GLM-5-Turbo", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 131072}, "cost": {"input": 0, "output": 0, "cache_read": 0, "cache_write": 0}}, "glm-image": {"name": "GLM-Image", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.015}, "id": "glm-image"}}}, "minimax": {"id": "minimax", "env": ["MINIMAX_API_KEY"], "npm": "@ai-sdk/anthropic", "api": "https://api.minimax.io/anthropic/v1", "name": "MiniMax (minimax.io)", "doc": "https://platform.minimax.io/docs/guides/quickstart", "models": {"MiniMax-M2": {"id": "MiniMax-M2", "name": "MiniMax-M2", "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 196608, "output": 128000}, "cost": {"input": 0.3, "output": 1.2}}, "MiniMax-M2.7": {"id": "MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.06, "cache_write": 0.375}}, "MiniMax-M2.1": {"id": "MiniMax-M2.1", "name": "MiniMax-M2.1", "description": "Earlier MiniMax agent model for practical coding and productivity tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375}}, "MiniMax-M2.5": {"id": "MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.03, "cache_write": 0.375}}, "MiniMax-M2.5-highspeed": {"id": "MiniMax-M2.5-highspeed", "name": "MiniMax-M2.5-highspeed", "description": "High-speed MiniMax model for low-latency coding and agent workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-02-13", "last_updated": "2026-02-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375}}, "MiniMax-M2.7-highspeed": {"id": "MiniMax-M2.7-highspeed", "name": "MiniMax-M2.7-highspeed", "description": "Low-latency M2.7 variant for interactive coding plans and agent loops", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.6, "output": 2.4, "cache_read": 0.06, "cache_write": 0.375}}, "MiniMax-M3": {"id": "MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-25", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.06, "tiers": [{"input": 0.6, "output": 2.4, "cache_read": 0.12, "tier": {"type": "context", "size": 512000}}], "context_over_200k": {"input": 0.6, "output": 2.4, "cache_read": 0.12}}}}}, "groq": {"id": "groq", "env": ["GROQ_API_KEY"], "npm": "@ai-sdk/groq", "name": "Groq", "doc": "https://console.groq.com/docs/models", "models": {"llama-3.3-70b-versatile": {"id": "llama-3.3-70b-versatile", "name": "Llama 3.3 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.59, "output": 0.79}}, "llama-3.1-8b-instant": {"id": "llama-3.1-8b-instant", "name": "Llama 3.1 8B", "description": "Compact Llama instruction model for fast chat and local deployment", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.05, "output": 0.08}}, "whisper-large-v3": {"id": "whisper-large-v3", "name": "Whisper", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2023-09-01", "last_updated": "2025-09-05", "modalities": {"input": ["audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 0, "output": 0}}, "allam-2-7b": {"id": "allam-2-7b", "name": "ALLaM-2-7b", "description": "ALLaM-2-7b instruction tuned model by SDAIA", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-01-23", "last_updated": "2025-01-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 4096, "output": 4096}, "cost": {"input": 0, "output": 0}}, "whisper-large-v3-turbo": {"id": "whisper-large-v3-turbo", "name": "Whisper Large V3 Turbo", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "whisper", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2024-10-01", "last_updated": "2024-10-01", "modalities": {"input": ["audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 0, "output": 0}}, "qwen/qwen3.6-27b": {"id": "qwen/qwen3.6-27b", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "default"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.6, "output": 3, "cache_read": 0.3}}, "canopylabs/orpheus-arabic-saudi": {"id": "canopylabs/orpheus-arabic-saudi", "name": "Canopy Labs Orpheus Arabic Saudi", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "canopylabs", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": {"input": ["text"], "output": ["audio"]}, "open_weights": false, "limit": {"context": 4000, "output": 50000}, "status": "beta"}, "canopylabs/orpheus-v1-english": {"id": "canopylabs/orpheus-v1-english", "name": "Canopy Labs Orpheus V1 English", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "canopylabs", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-12-19", "last_updated": "2025-12-19", "modalities": {"input": ["text"], "output": ["audio"]}, "open_weights": false, "limit": {"context": 4000, "output": 50000}, "status": "beta"}, "groq/compound-mini": {"id": "groq/compound-mini", "name": "Compound Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "groq", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-04", "last_updated": "2025-09-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 8192}}, "groq/compound": {"id": "groq/compound", "name": "Compound", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "groq", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2025-09-04", "last_updated": "2025-09-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 8192}}, "openai/gpt-oss-20b": {"id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-09-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 0.075, "output": 0.3, "cache_read": 0.0375}}, "openai/gpt-oss-safeguard-20b": {"id": "openai/gpt-oss-safeguard-20b", "name": "Safety GPT OSS 20B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-29", "last_updated": "2026-06-29", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 65536}, "status": "beta", "cost": {"input": 0.075, "output": 0.3}}, "openai/gpt-oss-120b": {"id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-10-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 0.15, "output": 0.6, "cache_read": 0.075}}, "meta-llama/llama-prompt-guard-2-22m": {"id": "meta-llama/llama-prompt-guard-2-22m", "name": "Llama Prompt Guard 2 22M", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-05-29", "last_updated": "2025-05-29", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 512, "output": 512}, "status": "beta", "cost": {"input": 0.03, "output": 0.03}}, "meta-llama/llama-prompt-guard-2-86m": {"id": "meta-llama/llama-prompt-guard-2-86m", "name": "Prompt Guard 2 86M", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "release_date": "2025-05-29", "last_updated": "2025-05-29", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 512, "output": 512}, "status": "beta", "cost": {"input": 0.04, "output": 0.04}}}}, "deepseek": {"id": "deepseek", "env": ["DEEPSEEK_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.deepseek.com", "name": "DeepSeek", "doc": "https://api-docs.deepseek.com/quick_start/pricing", "models": {"deepseek-v4-flash": {"id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-07-31", "last_updated": "2026-07-31", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 384000}, "cost": {"input": 0.14, "output": 0.28, "reasoning": 0.28, "cache_read": 0.0028}}, "deepseek-v4-pro": {"id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-08-12", "last_updated": "2026-08-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 384000}, "cost": {"input": 0.435, "output": 0.87, "reasoning": 0.87, "cache_read": 0.003625}}, "deepseek-chat": {"id": "deepseek-chat", "name": "DeepSeek Chat", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": true, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-12-01", "last_updated": "2026-02-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 384000}, "cost": {"input": 0.14, "output": 0.28, "cache_read": 0.0028}}, "deepseek-reasoner": {"id": "deepseek-reasoner", "name": "DeepSeek Reasoner", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2025-09", "release_date": "2025-12-01", "last_updated": "2026-02-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 384000}, "cost": {"input": 0.14, "output": 0.28, "reasoning": 0.28, "cache_read": 0.0028}}}}, "alibaba": {"id": "alibaba", "env": ["DASHSCOPE_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1", "name": "Alibaba", "doc": "https://www.alibabacloud.com/help/en/model-studio/models", "models": {"qwen3-coder-480b-a35b-instruct": {"id": "qwen3-coder-480b-a35b-instruct", "name": "Qwen3-Coder 480B-A35B Instruct", "description": "Open Qwen coding heavyweight for repository reasoning and agentic engineering", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 1.5, "output": 7.5}}, "qwen3.7-plus": {"id": "qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-04", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.625, "tiers": [{"input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5, "tier": {"type": "context", "size": 256000}}], "context_over_200k": {"input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5}}}, "qwen3-vl-plus": {"id": "qwen3-vl-plus", "name": "Qwen3-VL Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.2, "output": 1.6, "reasoning": 4.8}}, "qwen3-32b": {"id": "qwen3-32b", "name": "Qwen3 32B", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.7, "output": 2.8, "reasoning": 8.4}}, "qwen3.6-35b-a3b": {"id": "qwen3.6-35b-a3b", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": {"input": ["text", "image", "video", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.248, "output": 1.485}}, "qwen-mt-plus": {"id": "qwen-mt-plus", "name": "Qwen-MT Plus", "description": "Translation model for multilingual conversion, localization, and cross-language workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2025-01", "last_updated": "2025-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 16384, "output": 8192}, "cost": {"input": 2.46, "output": 7.37}}, "qwen2-5-vl-72b-instruct": {"id": "qwen2-5-vl-72b-instruct", "name": "Qwen2.5-VL 72B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 2.8, "output": 8.4}}, "qwen-max": {"id": "qwen-max", "name": "Qwen Max", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-04-03", "last_updated": "2025-01-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 32768, "output": 8192}, "cost": {"input": 1.6, "output": 6.4}}, "qwen3.5-plus": {"id": "qwen3.5-plus", "name": "Qwen3.5 Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.4, "output": 2.4, "reasoning": 2.4}}, "qwen-omni-turbo": {"id": "qwen-omni-turbo", "name": "Qwen-Omni Turbo", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-01-19", "last_updated": "2025-03-26", "modalities": {"input": ["text", "image", "audio", "video"], "output": ["text", "audio"]}, "open_weights": false, "limit": {"context": 32768, "output": 2048}, "cost": {"input": 0.07, "output": 0.27, "input_audio": 4.44, "output_audio": 8.89}}, "qwen-vl-max": {"id": "qwen-vl-max", "name": "Qwen-VL Max", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-04-08", "last_updated": "2025-08-13", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.8, "output": 3.2}}, "qwen3-coder-30b-a3b-instruct": {"id": "qwen3-coder-30b-a3b-instruct", "name": "Qwen3-Coder 30B-A3B Instruct", "description": "Smaller Qwen coder for efficient local agents and repo-level fixes", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.45, "output": 2.25}}, "qwen3.5-27b": {"id": "qwen3.5-27b", "name": "Qwen3.5 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image", "video", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.3, "output": 2.4}}, "qwen3-vl-235b-a22b": {"id": "qwen3-vl-235b-a22b", "name": "Qwen3-VL 235B-A22B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.7, "output": 2.8, "reasoning": 8.4}}, "qwen2-5-72b-instruct": {"id": "qwen2-5-72b-instruct", "name": "Qwen2.5 72B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 1.4, "output": 5.6}}, "glm-5.2": {"id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "minimal", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 131072}, "cost": {"input": 1.4, "output": 4.4, "cache_read": 0.28, "cache_write": 0}}, "qwen3.7-max": {"id": "qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 2.5, "output": 7.5, "cache_read": 0.5, "cache_write": 3.125}}, "qwen3-next-80b-a3b-thinking": {"id": "qwen3-next-80b-a3b-thinking", "name": "Qwen3-Next 80B-A3B (Thinking)", "description": "Efficient Qwen thinking model for local reasoning, math, and coding agents", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.5, "output": 6}}, "qwen-mt-turbo": {"id": "qwen-mt-turbo", "name": "Qwen-MT Turbo", "description": "Translation model for multilingual conversion, localization, and cross-language workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2025-01", "last_updated": "2025-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 16384, "output": 8192}, "cost": {"input": 0.16, "output": 0.49}}, "qvq-max": {"id": "qvq-max", "name": "QVQ Max", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qvq", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-03-25", "last_updated": "2025-03-25", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 1.2, "output": 4.8}}, "qwen2-5-vl-7b-instruct": {"id": "qwen2-5-vl-7b-instruct", "name": "Qwen2.5-VL 7B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.35, "output": 1.05}}, "qwen3.6-27b": {"id": "qwen3.6-27b", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": {"input": ["text", "image", "video", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.6, "output": 3.6}}, "qwen3.5-35b-a3b": {"id": "qwen3.5-35b-a3b", "name": "Qwen3.5 35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image", "video", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.25, "output": 2}}, "qwen3-vl-30b-a3b": {"id": "qwen3-vl-30b-a3b", "name": "Qwen3-VL 30B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.2, "output": 0.8, "reasoning": 2.4}}, "qwen3-14b": {"id": "qwen3-14b", "name": "Qwen3 14B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.35, "output": 1.4, "reasoning": 4.2}}, "qwen2-5-32b-instruct": {"id": "qwen2-5-32b-instruct", "name": "Qwen2.5 32B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.7, "output": 2.8}}, "qwen3-omni-flash-realtime": {"id": "qwen3-omni-flash-realtime", "name": "Qwen3-Omni Flash Realtime", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": {"input": ["text", "image", "audio", "video"], "output": ["text", "audio"]}, "open_weights": false, "limit": {"context": 65536, "output": 16384}, "cost": {"input": 0.52, "output": 1.99, "input_audio": 4.57, "output_audio": 18.13}}, "qwen3-235b-a22b": {"id": "qwen3-235b-a22b", "name": "Qwen3 235B-A22B", "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.7, "output": 2.8, "reasoning": 8.4}}, "qwen3-max": {"id": "qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 1.2, "output": 6}}, "qwen3-8b": {"id": "qwen3-8b", "name": "Qwen3 8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.18, "output": 0.7, "reasoning": 2.1}}, "qwen2-5-14b-instruct": {"id": "qwen2-5-14b-instruct", "name": "Qwen2.5 14B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.35, "output": 1.4}}, "qwen3-coder-plus": {"id": "qwen3-coder-plus", "name": "Qwen3 Coder Plus", "description": "Hosted Qwen coder for software agents, repo edits, and long-context code", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 1, "output": 5}}, "qwen2-5-7b-instruct": {"id": "qwen2-5-7b-instruct", "name": "Qwen2.5 7B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-09", "last_updated": "2024-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.175, "output": 0.7}}, "qwen3.8-max": {"id": "qwen3.8-max", "name": "Qwen3.8 Max", "description": "2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "xhigh"]}, {"type": "budget_tokens", "min": 0, "max": 262144}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-08-03", "last_updated": "2026-08-03", "modalities": {"input": ["text", "image", "video", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 131072}, "cost": {"input": 2, "output": 6, "cache_read": 0.25, "cache_write": 2.5}}, "qwen3-coder-flash": {"id": "qwen3-coder-flash", "name": "Qwen3 Coder Flash", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.3, "output": 1.5}}, "deepseek-v4-flash-0731": {"id": "deepseek-v4-flash-0731", "name": "DeepSeek V4 Flash 0731", "description": "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-07-31", "last_updated": "2026-07-31", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 384000}, "cost": {"input": 0.2, "output": 0.4, "cache_read": 0.04}}, "qwen-flash": {"id": "qwen-flash", "name": "Qwen Flash", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 32768}, "cost": {"input": 0.05, "output": 0.4}}, "qwen-plus": {"id": "qwen-plus", "name": "Qwen Plus", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01-25", "last_updated": "2025-09-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 32768}, "cost": {"input": 0.4, "output": 1.2, "reasoning": 4}}, "qwen-omni-turbo-realtime": {"id": "qwen-omni-turbo-realtime", "name": "Qwen-Omni Turbo Realtime", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-05-08", "last_updated": "2025-05-08", "modalities": {"input": ["text", "image", "audio"], "output": ["text", "audio"]}, "open_weights": false, "limit": {"context": 32768, "output": 2048}, "cost": {"input": 0.27, "output": 1.07, "input_audio": 4.44, "output_audio": 8.89}}, "qwen-turbo": {"id": "qwen-turbo", "name": "Qwen Turbo", "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-11-01", "last_updated": "2025-04-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 16384}, "cost": {"input": 0.05, "output": 0.2, "reasoning": 0.5}}, "qwen2-5-omni-7b": {"id": "qwen2-5-omni-7b", "name": "Qwen2.5-Omni 7B", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-12", "last_updated": "2024-12", "modalities": {"input": ["text", "image", "audio", "video"], "output": ["text", "audio"]}, "open_weights": true, "limit": {"context": 32768, "output": 2048}, "cost": {"input": 0.1, "output": 0.4, "input_audio": 6.76}}, "qwen3-asr-flash": {"id": "qwen3-asr-flash", "name": "Qwen3-ASR Flash", "description": "Speech transcription model for accurate audio-to-text and captioning workflows", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-04", "release_date": "2025-09-08", "last_updated": "2025-09-08", "modalities": {"input": ["audio"], "output": ["text"]}, "open_weights": false, "limit": {"context": 53248, "output": 4096}, "cost": {"input": 0.035, "output": 0.035}}, "qwq-plus": {"id": "qwq-plus", "name": "QwQ Plus", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-03-05", "last_updated": "2025-03-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.8, "output": 2.4}}, "qwen3.6-flash": {"id": "qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.1875, "output": 1.125, "cache_write": 0.234375}}, "qwen3.5-397b-a17b": {"id": "qwen3.5-397b-a17b", "name": "Qwen3.5 397B-A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": {"input": ["text", "image", "video", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.6, "output": 3.6}}, "qwen3.6-max-preview": {"id": "qwen3.6-max-preview", "name": "Qwen3.6 Max Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 1.3, "output": 7.8, "cache_read": 0.13, "cache_write": 1.625}}, "qwen-vl-ocr": {"id": "qwen-vl-ocr", "name": "Qwen-VL OCR", "description": "OCR model for extracting structured text from documents and screenshots", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2024-10-28", "last_updated": "2025-04-13", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 34096, "output": 4096}, "cost": {"input": 0.72, "output": 0.72}}, "qwen3-livetranslate-flash-realtime": {"id": "qwen3-livetranslate-flash-realtime", "name": "Qwen3-LiveTranslate Flash Realtime", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": {"input": ["text", "image", "audio", "video"], "output": ["text", "audio"]}, "open_weights": false, "limit": {"context": 53248, "output": 4096}, "cost": {"input": 10, "output": 10, "input_audio": 10, "output_audio": 38}}, "qwen-plus-character-ja": {"id": "qwen-plus-character-ja", "name": "Qwen Plus Character (Japanese)", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01", "last_updated": "2024-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 8192, "output": 512}, "cost": {"input": 0.5, "output": 1.4}}, "qwen3-omni-flash": {"id": "qwen3-omni-flash", "name": "Qwen3-Omni Flash", "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-09-15", "last_updated": "2025-09-15", "modalities": {"input": ["text", "image", "audio", "video"], "output": ["text", "audio"]}, "open_weights": false, "limit": {"context": 65536, "output": 16384}, "cost": {"input": 0.43, "output": 1.66, "input_audio": 3.81, "output_audio": 15.11}}, "qwen3.6-plus": {"id": "qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.5, "output": 3, "cache_read": 0.05, "cache_write": 0.625, "tiers": [{"input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5, "tier": {"type": "context", "size": 256000}}], "context_over_200k": {"input": 2, "output": 6, "cache_read": 0.2, "cache_write": 2.5}}}, "qwen3.5-122b-a10b": {"id": "qwen3.5-122b-a10b", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image", "video", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.4, "output": 3.2}}, "qwen3-next-80b-a3b-instruct": {"id": "qwen3-next-80b-a3b-instruct", "name": "Qwen3-Next 80B-A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.5, "output": 2}}, "qwen-vl-plus": {"id": "qwen-vl-plus", "name": "Qwen-VL Plus", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01-25", "last_updated": "2025-08-15", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.21, "output": 0.63}}}}, "openrouter": {"id": "openrouter", "env": ["OPENROUTER_API_KEY"], "npm": "@openrouter/ai-sdk-provider", "api": "https://openrouter.ai/api/v1", "name": "OpenRouter", "doc": "https://openrouter.ai/models", "models": {"~openai/gpt-latest": {"id": "~openai/gpt-latest", "name": "OpenAI GPT Latest", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": {"input": ["pdf", "image", "text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "output": 128000}, "cost": {"input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25, "tiers": [{"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5}}}, "~openai/gpt-mini-latest": {"id": "~openai/gpt-mini-latest", "name": "OpenAI GPT Mini Latest", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": {"input": ["pdf", "image", "text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "output": 128000}, "cost": {"input": 0.75, "output": 4.5, "cache_read": 0.075}}, "microsoft/phi-4": {"id": "microsoft/phi-4", "name": "Phi 4", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "phi", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2025-01-10", "last_updated": "2025-01-10", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 16384, "output": 16384}, "cost": {"input": 0.07, "output": 0.14}}, "microsoft/wizardlm-2-8x22b": {"id": "microsoft/wizardlm-2-8x22b", "name": "WizardLM-2 8x22B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-04-30", "release_date": "2024-04-16", "last_updated": "2024-04-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 65535, "output": 8000}, "cost": {"input": 0.62, "output": 0.62}}, "cohere/command-r-08-2024": {"id": "cohere/command-r-08-2024", "name": "Command R", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4000}, "cost": {"input": 0.15, "output": 0.6}}, "cohere/command-a": {"id": "cohere/command-a", "name": "Command A", "description": "Cohere command model for multilingual enterprise agents, tools, and chat", "family": "command-a", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 8192}, "cost": {"input": 2.5, "output": 10}}, "cohere/command-r-plus-08-2024": {"id": "cohere/command-r-plus-08-2024", "name": "Command R+", "description": "Cohere's RAG workhorse for long-context enterprise search and tool use", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-08-30", "last_updated": "2024-08-30", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4000}, "cost": {"input": 2.5, "output": 10}}, "cohere/command-r7b-12-2024": {"id": "cohere/command-r7b-12-2024", "name": "Command R7B", "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows", "family": "command-r", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-01", "release_date": "2024-12-02", "last_updated": "2024-12-02", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 4000}, "cost": {"input": 0.0375, "output": 0.15}}, "cohere/north-mini-code:free": {"id": "cohere/north-mini-code:free", "name": "North Mini Code (free)", "description": "Cohere coding model for practical software engineering and agentic edits", "family": "north", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-06-17", "last_updated": "2026-06-17", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 64000}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3-ultra-550b-a55b:free": {"id": "nvidia/nemotron-3-ultra-550b-a55b:free", "name": "Nemotron 3 Ultra (free)", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["medium", "high"]}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3-nano-30b-a3b": {"id": "nvidia/nemotron-3-nano-30b-a3b", "name": "Nemotron 3 Nano 30B A3B", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-15", "last_updated": "2025-12-15", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 228000}, "cost": {"input": 0.05, "output": 0.2, "cache_read": 0.025}}, "nvidia/nemotron-nano-9b-v2:free": {"id": "nvidia/nemotron-nano-9b-v2:free", "name": "Nemotron Nano 9B V2 (free)", "description": "Compact Nemotron model for efficient reasoning and deployable AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-18", "last_updated": "2025-08-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3-super-120b-a12b": {"id": "nvidia/nemotron-3-super-120b-a12b", "name": "Nemotron 3 Super 120B A12B", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium"]}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 16384}, "cost": {"input": 0.085, "output": 0.4}}, "nvidia/nemotron-3-nano-30b-a3b:free": {"id": "nvidia/nemotron-3-nano-30b-a3b:free", "name": "Nemotron 3 Nano 30B A3B (free)", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-15", "last_updated": "2025-12-15", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 256000}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free": {"id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", "name": "Nemotron 3 Nano Omni (free)", "description": "Open Nemotron omni model combining reasoning with text, vision, and audio", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens"}], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-28", "last_updated": "2026-04-28", "modalities": {"input": ["text", "image", "video", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 65536}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3.5-lightning": {"id": "nvidia/nemotron-3.5-lightning", "name": "Nemotron 3.5 Lightning 30B A3B", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-08-11", "last_updated": "2026-08-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 262144}, "cost": {"input": 0.1, "output": 0.25, "cache_read": 0.05}}, "nvidia/nemotron-3.5-content-safety:free": {"id": "nvidia/nemotron-3.5-content-safety:free", "name": "Nemotron 3.5 Content Safety (free)", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3-ultra-550b-a55b": {"id": "nvidia/nemotron-3-ultra-550b-a55b", "name": "Nemotron 3 Ultra 550B A55B", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["medium", "high"]}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 512288, "output": 16384}, "cost": {"input": 0.6, "output": 3.6, "cache_read": 0.2}}, "nvidia/nemotron-3.5-lightning:free": {"id": "nvidia/nemotron-3.5-lightning:free", "name": "Nemotron 3.5 Lightning (free)", "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-08-11", "last_updated": "2026-08-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-nano-12b-v2-vl:free": {"id": "nvidia/nemotron-nano-12b-v2-vl:free", "name": "Nemotron Nano 12B 2 VL (free)", "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows", "family": "nemotron", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-10-28", "last_updated": "2025-10-28", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 0, "output": 0}}, "nvidia/nemotron-3-super-120b-a12b:free": {"id": "nvidia/nemotron-3-super-120b-a12b:free", "name": "Nemotron 3 Super (free)", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium"]}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0, "output": 0}}, "deepcogito/cogito-v2.1-671b": {"id": "deepcogito/cogito-v2.1-671b", "name": "Cogito v2.1 671B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "cogito", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 1.25, "output": 1.25}}, "google/lyria-3-pro-preview": {"name": "Lyria 3 Pro Preview", "release_date": "2026-05-14", "modalities": {"input": ["text"], "output": ["audio"]}, "cost": {"input": 0.08, "output": 0.0, "type": "request"}, "id": "google/lyria-3-pro-preview"}, "google/gemini-3.5-flash": {"id": "google/gemini-3.5-flash", "name": "Gemini 3.5 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-19", "last_updated": "2026-05-19", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 1.5, "output": 9, "reasoning": 9, "cache_read": 0.15, "cache_write": 0.083333}}, "google/gemini-2.5-flash": {"id": "google/gemini-2.5-flash", "name": "Gemini 2.5 Flash", "description": "Fast Gemini workhorse for multimodal apps where latency and price matter", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 0, "max": 24576}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": {"input": ["text", "image", "audio", "video", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65535}, "cost": {"input": 0.3, "output": 2.5, "reasoning": 2.5, "cache_read": 0.03, "cache_write": 0.083333}}, "google/gemma-3-4b-it": {"id": "google/gemma-3-4b-it", "name": "Gemma 3 4B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.05, "output": 0.1}}, "google/gemini-3.5-flash-lite": {"id": "google/gemini-3.5-flash-lite", "name": "Gemini 3.5 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-03", "release_date": "2026-07-21", "last_updated": "2026-07-21", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0.3, "output": 2.5, "reasoning": 2.5, "cache_read": 0.03, "cache_write": 0.083333}}, "google/gemini-2.5-pro-preview": {"id": "google/gemini-2.5-pro-preview", "name": "Gemini 2.5 Pro Preview 06-05", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 128, "max": 32768}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-06-05", "last_updated": "2025-06-05", "modalities": {"input": ["pdf", "image", "text", "audio"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 1.25, "output": 10, "reasoning": 10, "cache_read": 0.125, "cache_write": 0.375, "tiers": [{"input": 2.5, "output": 15, "cache_read": 0.25, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 15, "cache_read": 0.25}}}, "google/lyria-3-clip-preview": {"name": "Lyria 3 Clip Preview", "release_date": "2026-05-14", "modalities": {"input": ["text"], "output": ["audio"]}, "cost": {"input": 0.04, "output": 0.0, "type": "request"}, "id": "google/lyria-3-clip-preview"}, "google/gemini-3-pro-image-preview": {"name": "Google: Nano Banana Pro Preview (Gemini 3 Pro)", "release_date": "2025-11-20", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 2.0, "output": 12.0}, "id": "google/gemini-3-pro-image-preview"}, "google/gemini-3.1-flash-lite-preview": {"id": "google/gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0.25, "output": 1.5, "reasoning": 1.5, "cache_read": 0.025, "cache_write": 0.083333}}, "google/gemini-3-flash-preview": {"id": "google/gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-12-17", "last_updated": "2025-12-17", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0.5, "output": 3, "reasoning": 3, "cache_read": 0.05, "cache_write": 0.083333}}, "google/gemini-3.1-pro-preview-customtools": {"id": "google/gemini-3.1-pro-preview-customtools", "name": "Gemini 3.1 Pro Preview Custom Tools", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 2, "output": 12, "reasoning": 12, "cache_read": 0.2, "cache_write": 0.375, "tiers": [{"input": 4, "output": 18, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 18, "cache_read": 0.4}}}, "google/gemini-3.1-flash-lite-image": {"id": "google/gemini-3.1-flash-lite-image", "name": "Nano Banana 2 Lite", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["minimal", "high"]}], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": {"input": ["text", "image"], "output": ["text", "image"]}, "open_weights": false, "limit": {"context": 65536, "output": 65536}, "cost": {"input": 0.25, "output": 1.5}}, "google/gemini-3.1-flash-image-preview": {"name": "Gemini 3.1 Flash Image (Nano Banana 2)", "release_date": "2026-02-26", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0.25, "output": 1.5}, "id": "google/gemini-3.1-flash-image-preview"}, "google/gemma-4-26b-a4b-it": {"id": "google/gemma-4-26b-a4b-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["image", "text", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.12, "output": 0.4, "cache_read": 0.05}}, "google/gemma-2-27b-it": {"id": "google/gemma-2-27b-it", "name": "Gemma 2 27B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2024-07-13", "last_updated": "2024-07-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 8192, "output": 2048}, "cost": {"input": 0.65, "output": 0.65}}, "google/gemma-3-27b-it": {"id": "google/gemma-3-27b-it", "name": "Gemma 3 27B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-03-12", "last_updated": "2025-03-12", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 131072}, "cost": {"input": 0.08, "output": 0.45, "cache_read": 0.04}}, "google/gemma-4-26b-a4b-it:free": {"id": "google/gemma-4-26b-a4b-it:free", "name": "Gemma 4 26B A4B (free)", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["image", "text", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0, "output": 0}}, "google/gemini-3.6-flash": {"id": "google/gemini-3.6-flash", "name": "Gemini 3.6 Flash", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-03", "release_date": "2026-07-21", "last_updated": "2026-07-21", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 1.5, "output": 7.5, "reasoning": 7.5, "cache_read": 0.15, "cache_write": 0.083333}}, "google/gemini-3.1-flash-lite": {"id": "google/gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash Lite", "description": "Low-latency Gemini model for high-volume multimodal and agent workloads", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-07", "last_updated": "2026-05-07", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 0.25, "output": 1.5, "reasoning": 1.5, "cache_read": 0.025, "cache_write": 0.083333}}, "google/gemini-2.5-pro-preview-05-06": {"id": "google/gemini-2.5-pro-preview-05-06", "name": "Gemini 2.5 Pro Preview 05-06", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 128, "max": 32768}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": {"input": ["text", "image", "pdf", "audio", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65535}, "cost": {"input": 1.25, "output": 10, "reasoning": 10, "cache_read": 0.125, "cache_write": 0.375, "tiers": [{"input": 2.5, "output": 15, "cache_read": 0.25, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 15, "cache_read": 0.25}}}, "google/gemma-3-12b-it": {"id": "google/gemma-3-12b-it", "name": "Gemma 3 12B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-03-13", "last_updated": "2025-03-13", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.05, "output": 0.15}}, "google/gemma-4-31b-it": {"id": "google/gemma-4-31b-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["image", "text", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.1, "output": 0.34, "cache_read": 0.1}}, "google/gemini-3.1-flash-image": {"id": "google/gemini-3.1-flash-image", "name": "Nano Banana 2", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["minimal", "high"]}], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": {"input": ["image", "text"], "output": ["text", "image"]}, "open_weights": false, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.5, "output": 3}}, "google/gemini-2.5-flash-image": {"name": "Gemini 2.5 Flash Image (Nano Banana)", "release_date": "2025-10-07", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0.3, "output": 2.5}, "id": "google/gemini-2.5-flash-image"}, "google/gemma-3n-e4b-it": {"id": "google/gemma-3n-e4b-it", "name": "Gemma 3n 4B", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-05-20", "last_updated": "2025-05-20", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 32768}, "cost": {"input": 0.06, "output": 0.12}}, "google/gemini-3-pro-image": {"id": "google/gemini-3-pro-image", "name": "Nano Banana Pro", "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": {"input": ["text", "image"], "output": ["text", "image"]}, "open_weights": false, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 2, "output": 12, "reasoning": 12, "cache_read": 0.2, "cache_write": 0.375}}, "google/gemini-3.1-pro-preview": {"id": "google/gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-02-19", "last_updated": "2026-02-19", "modalities": {"input": ["text", "image", "video", "audio", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 2, "output": 12, "reasoning": 12, "cache_read": 0.2, "cache_write": 0.375, "tiers": [{"input": 4, "output": 18, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 18, "cache_read": 0.4}}}, "google/gemini-2.5-pro": {"id": "google/gemini-2.5-pro", "name": "Gemini 2.5 Pro", "description": "Google's proven reasoning model for coding, math, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 128, "max": 32768}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": {"input": ["text", "image", "audio", "video", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 1.25, "output": 10, "reasoning": 10, "cache_read": 0.125, "cache_write": 0.375, "tiers": [{"input": 2.5, "output": 15, "cache_read": 0.25, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 15, "cache_read": 0.25}}}, "google/gemini-2.5-flash-lite": {"id": "google/gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", "family": "gemini-flash-lite", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 512, "max": 24576}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": {"input": ["text", "image", "audio", "video", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65535}, "cost": {"input": 0.1, "output": 0.4, "reasoning": 0.4, "cache_read": 0.01, "cache_write": 0.083333}}, "google/gemma-4-31b-it:free": {"id": "google/gemma-4-31b-it:free", "name": "Gemma 4 31B (free)", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["image", "text", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0, "output": 0}}, "thinkingmachines/inkling-small": {"id": "thinkingmachines/inkling-small", "name": "Inkling Small", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "ling", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "minimal", "low", "medium", "high", "max"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-30", "last_updated": "2026-07-30", "modalities": {"input": ["text", "image", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 524288, "output": 262144}, "cost": {"input": 0.45, "output": 1.2, "cache_read": 0.1}}, "thinkingmachines/inkling": {"id": "thinkingmachines/inkling", "name": "Inkling", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "ling", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "minimal", "low", "medium", "high", "max"]}], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-15", "last_updated": "2026-07-15", "modalities": {"input": ["text", "image", "audio"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 262144}, "cost": {"input": 0.95, "output": 4.05, "cache_read": 0.16}}, "relace/relace-apply-3": {"id": "relace/relace-apply-3", "name": "Relace Apply 3", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "release_date": "2025-09-26", "last_updated": "2025-09-26", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 256000, "output": 128000}, "cost": {"input": 0.85, "output": 1.25}}, "relace/relace-search": {"id": "relace/relace-search", "name": "Relace Search", "description": "Tool-capable chat model for instruction following and agentic application workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 256000, "output": 128000}, "cost": {"input": 1, "output": 3}}, "~deepseek/deepseek-v4-flash-latest": {"id": "~deepseek/deepseek-v4-flash-latest", "name": "DeepSeek V4 Flash Latest", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "high", "max"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-08-01", "last_updated": "2026-08-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 262144}, "cost": {"input": 0.079996, "output": 0.252, "cache_read": 0.0252}}, "perceptron/perceptron-mk1": {"id": "perceptron/perceptron-mk1", "name": "Perceptron Mk1", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2026-05-12", "last_updated": "2026-05-12", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 32768, "output": 8192}, "cost": {"input": 0.15, "output": 1.5}}, "sakana/fugu-ultra": {"id": "sakana/fugu-ultra", "name": "Fugu Ultra", "description": "Quality-first multi-agent model for hard research, analysis, and competitions", "family": "fugu", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-06-15", "last_updated": "2026-06-15", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 5, "output": 30, "cache_read": 0.5, "tiers": [{"input": 10, "output": 45, "cache_read": 1, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 10, "output": 45, "cache_read": 1}}}, "sakana/sakana-namazu": {"id": "sakana/sakana-namazu", "name": "Sakana Namazu", "description": "Multi-agent model for routing expert agents across complex analytical tasks", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-08-11", "last_updated": "2026-08-11", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.95, "output": 4, "cache_read": 0.15}}, "~google/gemini-flash-latest": {"id": "~google/gemini-flash-latest", "name": "Google Gemini Flash Latest", "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost", "family": "gemini-flash", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01-01", "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": {"input": ["text", "image", "video", "pdf", "audio"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 1.5, "output": 7.5, "reasoning": 7.5, "cache_read": 0.15, "cache_write": 0.083333}}, "~google/gemini-pro-latest": {"id": "~google/gemini-pro-latest", "name": "Google Gemini Pro Latest", "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", "family": "gemini-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": {"input": ["audio", "pdf", "image", "text", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 65536}, "cost": {"input": 2, "output": 12, "reasoning": 12, "cache_read": 0.2, "cache_write": 0.375, "tiers": [{"input": 4, "output": 18, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 18, "cache_read": 0.4}}}, "qwen/qwen3.5-flash-02-23": {"id": "qwen/qwen3.5-flash-02-23", "name": "Qwen3.5-Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1, "max": 81920}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-25", "last_updated": "2026-02-25", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.065, "output": 0.26}}, "qwen/qwen3.7-flash": {"id": "qwen/qwen3.7-flash", "name": "Qwen3.7 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens"}], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-15", "last_updated": "2026-07-15", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "input": 991000, "output": 65536}, "cost": {"input": 0.03, "output": 0.13, "cache_read": 0.006, "cache_write": 0.038, "tiers": [{"input": 0.1, "output": 0.4, "cache_read": 0.02, "cache_write": 0.125, "tier": {"type": "context", "size": 32000}}, {"input": 0.2, "output": 0.8, "cache_read": 0.04, "cache_write": 0.25, "tier": {"type": "context", "size": 256000}}]}}, "qwen/qwen3.7-plus": {"id": "qwen/qwen3.7-plus", "name": "Qwen3.7 Plus", "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1, "max": 262144}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-06-02", "last_updated": "2026-06-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 131072}, "cost": {"input": 0.32, "output": 1.28, "cache_read": 0.064, "cache_write": 0.4, "tiers": [{"input": 0.96, "output": 3.84, "cache_read": 0.192, "cache_write": 1.2, "tier": {"type": "context", "size": 256000}}], "context_over_200k": {"input": 0.96, "output": 3.84, "cache_read": 0.192, "cache_write": 1.2}}}, "qwen/qwen3-32b": {"id": "qwen/qwen3-32b", "name": "Qwen3 32B", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.08, "output": 0.28}}, "qwen/qwen3.6-35b-a3b": {"id": "qwen/qwen3.6-35b-a3b", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.15, "output": 1, "cache_read": 0.05}}, "qwen/qwen3-30b-a3b": {"id": "qwen/qwen3-30b-a3b", "name": "Qwen3 30B A3B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.12, "output": 0.5}}, "qwen/qwen3-235b-a22b-thinking-2507": {"id": "qwen/qwen3-235b-a22b-thinking-2507", "name": "Qwen3 235B A22B Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.23, "output": 2.3}}, "qwen/qwen3-vl-30b-a3b-thinking": {"id": "qwen/qwen3-vl-30b-a3b-thinking", "name": "Qwen3 VL 30B A3B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.2, "output": 2.4}}, "qwen/qwen3-coder-30b-a3b-instruct": {"id": "qwen/qwen3-coder-30b-a3b-instruct", "name": "Qwen3-Coder 30B-A3B Instruct", "description": "Smaller Qwen coder for efficient local agents and repo-level fixes", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.07, "output": 0.28}}, "qwen/qwen3.5-plus-02-15": {"id": "qwen/qwen3.5-plus-02-15", "name": "Qwen3.5 Plus 2026-02-15", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1, "max": 81920}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-16", "last_updated": "2026-02-16", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.26, "output": 1.56, "tiers": [{"input": 0.325, "output": 1.95, "tier": {"type": "context", "size": 256000}}], "context_over_200k": {"input": 0.325, "output": 1.95}}}, "qwen/qwen3.5-27b": {"id": "qwen/qwen3.5-27b", "name": "Qwen3.5 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.195, "output": 1.56}}, "qwen/qwen3.8-2.4t-a95b": {"id": "qwen/qwen3.8-2.4t-a95b", "name": "Qwen3.8 2.4T A95B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-08-12", "last_updated": "2026-08-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 52429}, "cost": {"input": 2, "output": 6, "cache_read": 0.2}}, "qwen/qwen3.7-max": {"id": "qwen/qwen3.7-max", "name": "Qwen3.7 Max", "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1, "max": 262144}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-05-21", "last_updated": "2026-05-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 131072}, "cost": {"input": 1.475, "output": 4.425, "cache_read": 0.295, "cache_write": 1.84375}}, "qwen/qwen3-next-80b-a3b-thinking": {"id": "qwen/qwen3-next-80b-a3b-thinking", "name": "Qwen3-Next 80B-A3B (Thinking)", "description": "Efficient Qwen thinking model for local reasoning, math, and coding agents", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.15, "output": 1.2}}, "qwen/qwen3.5-9b": {"id": "qwen/qwen3.5-9b", "name": "Qwen3.5 9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.1, "output": 0.15}}, "qwen/qwen3.6-27b": {"id": "qwen/qwen3.6-27b", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.6, "output": 3.6, "cache_read": 0.12}}, "qwen/qwen3.5-35b-a3b": {"id": "qwen/qwen3.5-35b-a3b", "name": "Qwen3.5 35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.25, "output": 1.25, "cache_read": 0.25}}, "qwen/qwen-2.5-7b-instruct": {"id": "qwen/qwen-2.5-7b-instruct", "name": "Qwen2.5 7B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2024-10-16", "last_updated": "2024-10-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 32768}, "cost": {"input": 0.1, "output": 0.2}}, "qwen/qwen3-14b": {"id": "qwen/qwen3-14b", "name": "Qwen3 14B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.12, "output": 0.24}}, "qwen/qwen3-235b-a22b": {"id": "qwen/qwen3-235b-a22b", "name": "Qwen3 235B-A22B", "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1, "max": 38912}], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.455, "output": 1.82}}, "qwen/qwen3-max": {"id": "qwen/qwen3-max", "name": "Qwen3 Max", "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.78, "output": 3.9, "cache_read": 0.156, "cache_write": 0.975, "tiers": [{"input": 1.56, "output": 7.8, "cache_read": 0.312, "cache_write": 1.95, "tier": {"type": "context", "size": 32000}}, {"input": 1.95, "output": 9.75, "cache_read": 0.39, "cache_write": 2.4375, "tier": {"type": "context", "size": 128000}}]}}, "qwen/qwen3-vl-8b-instruct": {"id": "qwen/qwen3-vl-8b-instruct", "name": "Qwen3 VL 8B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-14", "last_updated": "2025-10-14", "modalities": {"input": ["image", "text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.117, "output": 0.455}}, "qwen/qwen3-8b": {"id": "qwen/qwen3-8b", "name": "Qwen3 8B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-04-28", "last_updated": "2025-04-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.117, "output": 0.455}}, "qwen/qwen3-coder-plus": {"id": "qwen/qwen3-coder-plus", "name": "Qwen3 Coder Plus", "description": "Hosted Qwen coder for software agents, repo edits, and long-context code", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.65, "output": 3.25, "cache_read": 0.13, "cache_write": 0.8125, "tiers": [{"input": 1.17, "output": 5.85, "cache_read": 0.234, "cache_write": 1.4625, "tier": {"type": "context", "size": 32000}}, {"input": 1.95, "output": 9.75, "cache_read": 0.39, "cache_write": 2.4375, "tier": {"type": "context", "size": 128000}}]}}, "qwen/qwen3.8-max": {"id": "qwen/qwen3.8-max", "name": "Qwen3.8 Max", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-08-03", "last_updated": "2026-08-03", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 131072}, "cost": {"input": 2, "output": 6, "cache_read": 0.25, "cache_write": 2.5}}, "qwen/qwen3-coder-flash": {"id": "qwen/qwen3-coder-flash", "name": "Qwen3 Coder Flash", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.195, "output": 0.975, "cache_read": 0.039, "cache_write": 0.24375, "tiers": [{"input": 0.325, "output": 1.625, "cache_read": 0.065, "cache_write": 0.40625, "tier": {"type": "context", "size": 32000}}, {"input": 0.52, "output": 2.6, "cache_read": 0.104, "cache_write": 0.65, "tier": {"type": "context", "size": 128000}}]}}, "qwen/qwen3-vl-235b-a22b-instruct": {"id": "qwen/qwen3-vl-235b-a22b-instruct", "name": "Qwen3 VL 235B A22B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.26, "output": 1.04}}, "qwen/qwen3-vl-32b-instruct": {"id": "qwen/qwen3-vl-32b-instruct", "name": "Qwen3 VL 32B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-23", "last_updated": "2025-10-23", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.104, "output": 0.416}}, "qwen/qwen-2.5-72b-instruct": {"id": "qwen/qwen-2.5-72b-instruct", "name": "Qwen2.5 72B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2024-09-19", "last_updated": "2024-09-19", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 16384}, "cost": {"input": 0.36, "output": 0.4}}, "qwen/qwen-plus-2025-07-28:thinking": {"id": "qwen/qwen-plus-2025-07-28:thinking", "name": "Qwen Plus 0728 (thinking)", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 1, "max": 81920}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-09-08", "last_updated": "2025-09-08", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 32768}, "cost": {"input": 0.4, "output": 1.2, "cache_write": 0.5, "tiers": [{"input": 1.2, "output": 3.6, "cache_write": 1.5, "tier": {"type": "context", "size": 256000}}], "context_over_200k": {"input": 1.2, "output": 3.6, "cache_write": 1.5}}}, "qwen/qwen-plus": {"id": "qwen/qwen-plus", "name": "Qwen Plus", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2024-01-25", "last_updated": "2025-09-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 32768}, "cost": {"input": 0.26, "output": 0.78, "cache_read": 0.052, "cache_write": 0.325, "tiers": [{"input": 0.78, "output": 2.34, "cache_read": 0.156, "cache_write": 0.975, "tier": {"type": "context", "size": 256000}}], "context_over_200k": {"input": 0.78, "output": 2.34, "cache_read": 0.156, "cache_write": 0.975}}}, "qwen/qwen3-30b-a3b-thinking-2507": {"id": "qwen/qwen3-30b-a3b-thinking-2507", "name": "Qwen3 30B A3B Thinking 2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-08-28", "last_updated": "2025-08-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 81920, "output": 32768}, "cost": {"input": 0.2, "output": 2.4}}, "qwen/qwen3-vl-8b-thinking": {"id": "qwen/qwen3-vl-8b-thinking", "name": "Qwen3 VL 8B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-14", "last_updated": "2025-10-14", "modalities": {"input": ["image", "text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.18, "output": 2.1}}, "qwen/qwen3-235b-a22b-2507": {"id": "qwen/qwen3-235b-a22b-2507", "name": "Qwen3 235B A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-07-21", "last_updated": "2025-07-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 16384}, "cost": {"input": 0.09, "output": 0.55}}, "qwen/qwen3.6-flash": {"id": "qwen/qwen3.6-flash", "name": "Qwen3.6 Flash", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.6", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1, "max": 81920}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.1875, "output": 1.125, "cache_write": 0.234375, "tiers": [{"input": 0.75, "output": 3, "cache_write": 0.9375, "tier": {"type": "context", "size": 256000}}], "context_over_200k": {"input": 0.75, "output": 3, "cache_write": 0.9375}}}, "qwen/qwen3-30b-a3b-instruct-2507": {"id": "qwen/qwen3-30b-a3b-instruct-2507", "name": "Qwen3 30B A3B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-07-29", "last_updated": "2025-07-29", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32000}, "cost": {"input": 0.04815, "output": 0.19305}}, "qwen/qwen3-vl-30b-a3b-instruct": {"id": "qwen/qwen3-vl-30b-a3b-instruct", "name": "Qwen3 VL 30B A3B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 16384}, "cost": {"input": 0.15, "output": 0.6}}, "qwen/qwen3.5-397b-a17b": {"id": "qwen/qwen3.5-397b-a17b", "name": "Qwen3.5 397B-A17B", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.5, "output": 3.6, "cache_read": 0.3}}, "qwen/qwen-2.5-coder-32b-instruct": {"id": "qwen/qwen-2.5-coder-32b-instruct", "name": "Qwen2.5 Coder 32B Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-06-30", "release_date": "2024-11-11", "last_updated": "2024-11-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 32768}, "cost": {"input": 0.66, "output": 1}}, "qwen/qwen3-vl-235b-a22b-thinking": {"id": "qwen/qwen3-vl-235b-a22b-thinking", "name": "Qwen3 VL 235B A22B Thinking", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-09-23", "last_updated": "2025-09-23", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.4, "output": 4}}, "qwen/qwen3.6-max-preview": {"id": "qwen/qwen3.6-max-preview", "name": "Qwen3.6 Max Preview", "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1, "max": 131072}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 1.027, "output": 6.162, "cache_write": 1.28375, "tiers": [{"input": 1.58, "output": 9.48, "cache_write": 1.975, "tier": {"type": "context", "size": 128000}}]}}, "qwen/qwen3-max-thinking": {"id": "qwen/qwen3-max-thinking", "name": "Qwen3 Max Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 1, "max": 81920}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-09", "last_updated": "2026-02-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.78, "output": 3.9, "tiers": [{"input": 1.56, "output": 7.8, "tier": {"type": "context", "size": 32000}}, {"input": 1.95, "output": 9.75, "tier": {"type": "context", "size": 128000}}]}}, "qwen/qwen3-coder": {"id": "qwen/qwen3-coder", "name": "Qwen3 Coder 480B A35B", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.3, "output": 1, "cache_read": 0.1}}, "qwen/qwen3-coder-next": {"id": "qwen/qwen3-coder-next", "name": "Qwen3 Coder Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-09", "release_date": "2026-02-03", "last_updated": "2026-02-03", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.12, "output": 0.8, "cache_read": 0.07}}, "qwen/qwen3.6-plus": {"id": "qwen/qwen3.6-plus", "name": "Qwen3.6 Plus", "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1, "max": 81920}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.325, "output": 1.95, "cache_write": 0.40625, "tiers": [{"input": 1.3, "output": 3.9, "cache_write": 1.625, "tier": {"type": "context", "size": 256000}}], "context_over_200k": {"input": 1.3, "output": 3.9, "cache_write": 1.625}}}, "qwen/qwen3.5-122b-a10b": {"id": "qwen/qwen3.5-122b-a10b", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 81920}, "cost": {"input": 0.29, "output": 2.4}}, "qwen/qwen3.5-plus-20260420": {"id": "qwen/qwen3.5-plus-20260420", "name": "Qwen3.5 Plus 2026-04-20", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen3.5", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1, "max": 81920}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65536}, "cost": {"input": 0.3, "output": 1.8, "cache_write": 0.375, "tiers": [{"input": 0.375, "output": 2.25, "cache_write": 0.46875, "tier": {"type": "context", "size": 256000}}], "context_over_200k": {"input": 0.375, "output": 2.25, "cache_write": 0.46875}}}, "qwen/qwen-plus-2025-07-28": {"id": "qwen/qwen-plus-2025-07-28", "name": "Qwen Plus 0728", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-09-08", "last_updated": "2025-09-08", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 32768}, "cost": {"input": 0.26, "output": 0.78, "tiers": [{"input": 0.78, "output": 2.34, "tier": {"type": "context", "size": 256000}}], "context_over_200k": {"input": 0.78, "output": 2.34}}}, "qwen/qwen3-next-80b-a3b-instruct": {"id": "qwen/qwen3-next-80b-a3b-instruct", "name": "Qwen3-Next 80B-A3B Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09", "last_updated": "2025-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 16384}, "cost": {"input": 0.09, "output": 1.1}}, "qwen/qwen2.5-vl-72b-instruct": {"id": "qwen/qwen2.5-vl-72b-instruct", "name": "Qwen2.5 VL 72B Instruct", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2025-02-01", "last_updated": "2025-02-01", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 0.25, "output": 0.75}}, "inception/mercury-2": {"id": "inception/mercury-2", "name": "Mercury 2", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "mercury", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-04", "last_updated": "2026-03-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 50000}, "cost": {"input": 0.25, "output": 0.75, "cache_read": 0.025}}, "rekaai/reka-edge": {"id": "rekaai/reka-edge", "name": "Reka Edge", "description": "Multimodal model for analyzing text, images, documents, and rich media", "family": "reka", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-20", "last_updated": "2026-03-20", "modalities": {"input": ["image", "text", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 16384, "output": 16384}, "cost": {"input": 0.1, "output": 0.1}}, "rekaai/reka-flash-3": {"id": "rekaai/reka-flash-3", "name": "Reka Flash 3", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "reka", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-03-12", "last_updated": "2025-03-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 65536, "output": 65536}, "cost": {"input": 0.1, "output": 0.2}}, "tencent/hunyuan-a13b-instruct": {"id": "tencent/hunyuan-a13b-instruct", "name": "Hunyuan A13B Instruct", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "hunyuan", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-07-08", "last_updated": "2025-07-08", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.14, "output": 0.57}}, "tencent/hy3": {"id": "tencent/hy3", "name": "Hy3", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "Hy", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-06", "last_updated": "2026-07-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 128000}, "cost": {"input": 0.132, "output": 0.528, "cache_read": 0.033}}, "tencent/hy3-preview": {"id": "tencent/hy3-preview", "name": "Hy3 preview", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "Hy", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "high"]}], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.063, "output": 0.21, "cache_read": 0.021}}, "upstage/solar-pro-3": {"id": "upstage/solar-pro-3", "name": "Solar Pro 3", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "solar-pro", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.15, "output": 0.6, "cache_read": 0.015}}, "upstage/solar-pro4": {"id": "upstage/solar-pro4", "name": "Solar Pro 4", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "solar", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-08-10", "last_updated": "2026-08-10", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 524288, "output": 131072}, "cost": {"input": 0.03, "output": 0.12, "cache_read": 0.006}}, "~x-ai/grok-latest": {"id": "~x-ai/grok-latest", "name": "Grok Latest", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 500000, "output": 1000000}, "cost": {"input": 2, "output": 6, "cache_read": 0.5, "tiers": [{"input": 4, "output": 12, "cache_read": 1, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 12, "cache_read": 1}}}, "mistralai/mistral-large": {"id": "mistralai/mistral-large", "name": "Mistral Large", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-11-30", "release_date": "2024-02-26", "last_updated": "2024-02-26", "modalities": {"input": ["text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 2, "output": 6, "cache_read": 0.2}}, "mistralai/ministral-8b-2512": {"id": "mistralai/ministral-8b-2512", "name": "Ministral 3 8B 2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.15, "output": 0.15, "cache_read": 0.015}}, "mistralai/ministral-14b-2512": {"id": "mistralai/ministral-14b-2512", "name": "Ministral 3 14B 2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.2, "output": 0.2, "cache_read": 0.02}}, "mistralai/ministral-3b-2512": {"id": "mistralai/ministral-3b-2512", "name": "Ministral 3 3B 2512", "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads", "family": "ministral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.1, "output": 0.1, "cache_read": 0.01}}, "mistralai/mistral-small-24b-instruct-2501": {"id": "mistralai/mistral-small-24b-instruct-2501", "name": "Mistral Small 3", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-10-31", "release_date": "2025-01-30", "last_updated": "2025-01-30", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 16384}, "cost": {"input": 0.05, "output": 0.08}}, "mistralai/mistral-large-2407": {"id": "mistralai/mistral-large-2407", "name": "Mistral Large 2407", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-03-31", "release_date": "2024-11-19", "last_updated": "2024-11-19", "modalities": {"input": ["text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 2, "output": 6, "cache_read": 0.2}}, "mistralai/mistral-medium-3.1": {"id": "mistralai/mistral-medium-3.1", "name": "Mistral Medium 3.1", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06-30", "release_date": "2025-08-13", "last_updated": "2025-08-13", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 262144}, "cost": {"input": 0.4, "output": 2, "cache_read": 0.04}}, "mistralai/mistral-nemo": {"id": "mistralai/mistral-nemo", "name": "Mistral Nemo", "description": "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.019, "output": 0.03}}, "mistralai/mistral-saba": {"id": "mistralai/mistral-saba", "name": "Saba", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-09-30", "release_date": "2025-02-17", "last_updated": "2025-02-17", "modalities": {"input": ["text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 32768, "output": 32768}, "cost": {"input": 0.2, "output": 0.6, "cache_read": 0.02}}, "mistralai/mistral-small-2603": {"id": "mistralai/mistral-small-2603", "name": "Mistral Small 4", "description": "Fast Mistral production model for chat, extraction, and cost-sensitive agents", "family": "mistral-small", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-06", "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.15, "output": 0.6, "cache_read": 0.015}}, "mistralai/codestral-2508": {"id": "mistralai/codestral-2508", "name": "Codestral 2508", "description": "Mistral coding model for code completion, generation, and developer workflows", "family": "codestral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-01", "last_updated": "2025-08-01", "modalities": {"input": ["text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 256000, "output": 256000}, "cost": {"input": 0.3, "output": 0.9, "cache_read": 0.03}}, "mistralai/mixtral-8x22b-instruct": {"id": "mistralai/mixtral-8x22b-instruct", "name": "Mixtral 8x22B Instruct", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-01-31", "release_date": "2024-04-17", "last_updated": "2024-04-17", "modalities": {"input": ["text", "pdf"], "output": ["text"]}, "open_weights": true, "limit": {"context": 65536, "output": 65536}, "cost": {"input": 2, "output": 6, "cache_read": 0.2}}, "mistralai/voxtral-small-24b-2507": {"id": "mistralai/voxtral-small-24b-2507", "name": "Voxtral Small 24B 2507", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-30", "last_updated": "2025-10-30", "modalities": {"input": ["text", "audio", "pdf"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32000, "output": 32000}, "cost": {"input": 0.1, "output": 0.3, "cache_read": 0.01}}, "mistralai/mistral-small-3.2-24b-instruct": {"id": "mistralai/mistral-small-3.2-24b-instruct", "name": "Mistral Small 3.2 24B", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-10-31", "release_date": "2025-06-20", "last_updated": "2025-06-20", "modalities": {"input": ["image", "text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 16384}, "cost": {"input": 0.09375, "output": 0.25}}, "mistralai/mistral-medium-3-5": {"id": "mistralai/mistral-medium-3-5", "name": "Mistral Medium 3.5", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 1.5, "output": 7.5}}, "mistralai/mistral-small-3.1-24b-instruct": {"id": "mistralai/mistral-small-3.1-24b-instruct", "name": "Mistral Small 3.1 24B", "description": "Efficient Mistral model for fast chat, extraction, and production assistants", "family": "mistral-small", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2023-10-31", "release_date": "2025-03-17", "last_updated": "2025-03-17", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 0.351, "output": 0.555}}, "mistralai/mistral-medium-3": {"id": "mistralai/mistral-medium-3", "name": "Mistral Medium 3", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral-medium", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-07", "last_updated": "2025-05-07", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.4, "output": 2, "cache_read": 0.04}}, "mistralai/mistral-large-2512": {"id": "mistralai/mistral-large-2512", "name": "Mistral Large 3", "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-11", "release_date": "2024-11-01", "last_updated": "2025-12-02", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.5, "output": 1.5, "cache_read": 0.05}}, "bytedance/ui-tars-1.5-7b": {"id": "bytedance/ui-tars-1.5-7b", "name": "UI-TARS 7B ", "description": "Multimodal model for analyzing text, images, documents, and rich media", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-07-22", "last_updated": "2025-07-22", "modalities": {"input": ["image", "text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 2048}, "cost": {"input": 0.1, "output": 0.2, "cache_read": 0.1}}, "nex-agi/nex-n2-mini": {"id": "nex-agi/nex-n2-mini", "name": "Nex-N2-Mini", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "agi", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-24", "last_updated": "2026-06-24", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.025, "output": 0.1, "cache_read": 0.0025}}, "nex-agi/nex-n2-pro": {"id": "nex-agi/nex-n2-pro", "name": "Nex-N2-Pro", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "agi", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-06-08", "last_updated": "2026-06-08", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.25, "output": 1, "cache_read": 0.025}}, "thedrummer/rocinante-12b": {"id": "thedrummer/rocinante-12b", "name": "Rocinante 12B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-04-30", "release_date": "2024-09-30", "last_updated": "2024-09-30", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 65536, "output": 65536}, "cost": {"input": 0.25, "output": 0.5}}, "thedrummer/unslopnemo-12b": {"id": "thedrummer/unslopnemo-12b", "name": "UnslopNemo 12B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04-30", "release_date": "2024-11-08", "last_updated": "2024-11-08", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1024000, "output": 1024000}, "cost": {"input": 0.4, "output": 0.4}}, "thedrummer/cydonia-24b-v4.1": {"id": "thedrummer/cydonia-24b-v4.1", "name": "Cydonia 24B V4.1", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-04-30", "release_date": "2025-09-27", "last_updated": "2025-09-27", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.3, "output": 0.5, "cache_read": 0.15}}, "thedrummer/skyfall-36b-v2": {"id": "thedrummer/skyfall-36b-v2", "name": "Skyfall 36B V2", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2025-03-10", "last_updated": "2025-03-10", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 32768}, "cost": {"input": 0.55, "output": 0.8, "cache_read": 0.25}}, "undi95/remm-slerp-l2-13b": {"id": "undi95/remm-slerp-l2-13b", "name": "ReMM SLERP 13B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-06-30", "release_date": "2023-07-22", "last_updated": "2023-07-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 6144, "output": 6144}, "cost": {"input": 0.45, "output": 0.65}}, "meta/muse-glimmer-30b": {"id": "meta/muse-glimmer-30b", "name": "Muse Glimmer 30B", "description": "Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.", "family": "muse", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-01-04", "release_date": "2026-08-10", "last_updated": "2026-08-10", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.35, "output": 1.5, "cache_read": 0.04}}, "meta/muse-spark-1.1": {"id": "meta/muse-spark-1.1", "name": "Muse Spark 1.1", "description": "Open Llama multimodal model for image understanding and text reasoning", "family": "muse", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-08", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "video", "pdf", "audio"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 1048576}, "cost": {"input": 1.25, "output": 4.25, "cache_read": 0.15}}, "meta/muse-spark-1.2": {"id": "meta/muse-spark-1.2", "name": "Muse Spark 1.2", "description": "Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.", "family": "muse", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-08-05", "last_updated": "2026-08-05", "modalities": {"input": ["text", "image", "video", "pdf", "audio"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 1048576}, "cost": {"input": 1.25, "output": 4.25, "cache_read": 0.15}}, "gryphe/mythomax-l2-13b": {"id": "gryphe/mythomax-l2-13b", "name": "MythoMax 13B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-06-30", "release_date": "2023-07-02", "last_updated": "2023-07-02", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 8192, "output": 4096}, "cost": {"input": 0.06, "output": 0.06}}, "inclusionai/ling-3.0-flash": {"id": "inclusionai/ling-3.0-flash", "name": "Ling-3.0-flash", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "ling", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-23", "last_updated": "2026-07-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.021, "output": 0.063, "cache_read": 0.0042}}, "inclusionai/ring-2.6-1t": {"id": "inclusionai/ring-2.6-1t", "name": "Ring-2.6-1T", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "ring", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high", "xhigh"]}], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-05-08", "last_updated": "2026-05-08", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.075, "output": 0.625, "cache_read": 0.015}}, "inclusionai/ling-2.6-1t": {"id": "inclusionai/ling-2.6-1t", "name": "Ling-2.6-1T", "description": "Tool-capable chat model for instruction following and agentic application workflows", "family": "ling", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.075, "output": 0.625, "cache_read": 0.015}}, "inclusionai/ling-2.6-flash": {"id": "inclusionai/ling-2.6-flash", "name": "Ling-2.6-flash", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "ling", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.01, "output": 0.03, "cache_read": 0.002}}, "cognitivecomputations/dolphin-mistral-24b-venice-edition": {"id": "cognitivecomputations/dolphin-mistral-24b-venice-edition", "name": "Uncensored", "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", "family": "mistral", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-04-30", "release_date": "2025-07-09", "last_updated": "2025-07-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 8192}, "cost": {"input": 0.2, "output": 0.9}}, "allenai/olmo-3-32b-think": {"id": "allenai/olmo-3-32b-think", "name": "Olmo 3 32B Think", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "allenai", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-11-21", "last_updated": "2025-11-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 65536, "output": 65536}, "cost": {"input": 0.15, "output": 0.5}}, "meituan/longcat-2.0": {"id": "meituan/longcat-2.0", "name": "LongCat 2.0", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "longcat", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "budget_tokens"}], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-20", "last_updated": "2026-07-20", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048756, "output": 262144}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.006}}, "ai21/jamba-large-1.7": {"id": "ai21/jamba-large-1.7", "name": "Jamba Large 1.7", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "jamba", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 256000, "output": 4096}, "cost": {"input": 2, "output": 8}}, "mancer/weaver": {"id": "mancer/weaver", "name": "Weaver (alpha)", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "alpha", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-06-30", "release_date": "2023-08-02", "last_updated": "2023-08-02", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 8000, "output": 6000}, "cost": {"input": 0.5, "output": 0.75}}, "morph/morph-v3-large": {"id": "morph/morph-v3-large", "name": "Morph V3 Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "morph", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-07-07", "last_updated": "2025-07-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 131072}, "cost": {"input": 0.9, "output": 1.9}}, "morph/morph-v3-fast": {"id": "morph/morph-v3-fast", "name": "Morph V3 Fast", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "morph", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-07-07", "last_updated": "2025-07-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 81920, "output": 38000}, "cost": {"input": 0.8, "output": 1.2}}, "nousresearch/hermes-3-llama-3.1-70b": {"id": "nousresearch/hermes-3-llama-3.1-70b", "name": "Hermes 3 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "nousresearch", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-08-18", "last_updated": "2024-08-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.7, "output": 0.7}}, "nousresearch/hermes-3-llama-3.1-405b": {"id": "nousresearch/hermes-3-llama-3.1-405b", "name": "Hermes 3 405B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "nousresearch", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-08-16", "last_updated": "2024-08-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 1, "output": 1}}, "nousresearch/hermes-4-405b": {"id": "nousresearch/hermes-4-405b", "name": "Hermes 4 405B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "hermes", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 1, "output": 3}}, "nousresearch/hermes-4-70b": {"id": "nousresearch/hermes-4-70b", "name": "Hermes 4 70B", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "hermes", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-08-26", "last_updated": "2025-08-26", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.13, "output": 0.4}}, "poolside/laguna-xs-2.1": {"id": "poolside/laguna-xs-2.1", "name": "Laguna XS 2.1", "description": "Agentic coding model from Poolside in the XS size class for local deployment", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-02", "last_updated": "2026-07-02", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.06, "output": 0.12, "cache_read": 0.03}}, "poolside/laguna-s-2.1": {"id": "poolside/laguna-s-2.1", "name": "Laguna S 2.1", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "family": "laguna-s", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-21", "last_updated": "2026-07-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 131072}, "cost": {"input": 0.09, "output": 0.18, "cache_read": 0.009}}, "poolside/laguna-s-2.1:free": {"id": "poolside/laguna-s-2.1:free", "name": "Laguna S 2.1 (free)", "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads", "family": "laguna-s", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-21", "last_updated": "2026-07-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0, "output": 0}}, "poolside/laguna-xs-2.1:free": {"id": "poolside/laguna-xs-2.1:free", "name": "Laguna XS 2.1 (free)", "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads", "family": "laguna", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-02", "last_updated": "2026-07-02", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0, "output": 0}}, "minimax/minimax-m2-her": {"id": "minimax/minimax-m2-her", "name": "MiniMax M2-her", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-01-23", "last_updated": "2026-01-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 65536, "output": 2048}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.03}}, "minimax/minimax-m2.7": {"id": "minimax/minimax-m2.7", "name": "MiniMax-M2.7", "description": "Open MiniMax flagship for coding agents, office automation, and complex environments", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.06}}, "minimax/minimax-m3": {"id": "minimax/minimax-m3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 512000}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.06}}, "minimax/minimax-m1": {"id": "minimax/minimax-m1", "name": "MiniMax M1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-06-30", "release_date": "2025-06-17", "last_updated": "2025-06-17", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 40000}, "cost": {"input": 0.55, "output": 2.2}}, "minimax/minimax-m2.5": {"id": "minimax/minimax-m2.5", "name": "MiniMax-M2.5", "description": "Prior MiniMax coding model for agent workflows, office edits, and automation", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 196608}, "cost": {"input": 0.22, "output": 0.9, "cache_read": 0.05}}, "minimax/minimax-m2": {"id": "minimax/minimax-m2", "name": "MiniMax-M2", "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": true, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.255, "output": 1.02}}, "minimax/minimax-01": {"id": "minimax/minimax-01", "name": "MiniMax-01", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-03-31", "release_date": "2025-01-15", "last_updated": "2025-01-15", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000192, "output": 1000192}, "cost": {"input": 0.2, "output": 1.1}}, "minimax/minimax-m2.1": {"id": "minimax/minimax-m2.1", "name": "MiniMax-M2.1", "description": "Earlier MiniMax agent model for practical coding and productivity tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": false, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.03}}, "deepseek/deepseek-v3.2-exp": {"id": "deepseek/deepseek-v3.2-exp", "name": "DeepSeek V3.2 Exp", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 163840, "output": 65536}, "cost": {"input": 0.27, "output": 0.41}}, "deepseek/deepseek-v4-flash": {"id": "deepseek/deepseek-v4-flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "xhigh"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 393216}, "cost": {"input": 0.14, "output": 0.28, "cache_read": 0.028}}, "deepseek/deepseek-chat-v3-0324": {"id": "deepseek/deepseek-chat-v3-0324", "name": "DeepSeek V3 0324", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07-31", "release_date": "2025-03-24", "last_updated": "2025-03-24", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 163840, "output": 65536}, "cost": {"input": 0.27, "output": 1.12, "cache_read": 0.135}}, "deepseek/deepseek-r1-distill-llama-70b": {"id": "deepseek/deepseek-r1-distill-llama-70b", "name": "R1 Distill Llama 70B", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-07-31", "release_date": "2025-01-23", "last_updated": "2025-01-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 8192, "output": 8192}, "cost": {"input": 0.8, "output": 0.8}}, "deepseek/deepseek-v4-pro": {"id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "xhigh"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 393216}, "cost": {"input": 1.168, "output": 2.336, "cache_read": 0.09855}}, "deepseek/deepseek-r1-0528": {"id": "deepseek/deepseek-r1-0528", "name": "R1 0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 163840, "output": 32768}, "cost": {"input": 0.5, "output": 2.15, "cache_read": 0.35}}, "deepseek/deepseek-v4-flash-0731": {"id": "deepseek/deepseek-v4-flash-0731", "name": "DeepSeek V4 Flash 0731", "description": "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "high", "max"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-07-31", "last_updated": "2026-07-31", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 384000}, "cost": {"input": 0.08, "output": 0.18, "cache_read": 0.016}}, "deepseek/deepseek-v3.2": {"id": "deepseek/deepseek-v3.2", "name": "DeepSeek V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 163840, "output": 65536}, "cost": {"input": 0.269, "output": 0.4, "cache_read": 0.1345}}, "deepseek/deepseek-r1": {"id": "deepseek/deepseek-r1", "name": "DeepSeek-R1", "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-05-29", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 163840, "output": 16000}, "cost": {"input": 0.7, "output": 2.5}}, "deepseek/deepseek-chat": {"id": "deepseek/deepseek-chat", "name": "DeepSeek Chat", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-09", "release_date": "2025-12-01", "last_updated": "2026-02-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 163840, "output": 16000}, "cost": {"input": 0.2574, "output": 1.0287}}, "deepseek/deepseek-v3.1-terminus": {"id": "deepseek/deepseek-v3.1-terminus", "name": "DeepSeek V3.1 Terminus", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-09-22", "last_updated": "2025-09-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 163840, "output": 32768}, "cost": {"input": 0.27, "output": 0.95, "cache_read": 0.13}}, "deepseek/deepseek-v4-pro-0813": {"id": "deepseek/deepseek-v4-pro-0813", "name": "DeepSeek V4 Pro 0813", "description": "DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "high", "max"]}], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-08-12", "last_updated": "2026-08-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 384000}, "cost": {"input": 0.435, "output": 0.87, "cache_read": 0.003625}}, "deepseek/deepseek-chat-v3.1": {"id": "deepseek/deepseek-chat-v3.1", "name": "DeepSeek V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 163840, "output": 32768}, "cost": {"input": 0.25, "output": 0.95, "cache_read": 0.13}}, "amazon/nova-premier-v1": {"id": "amazon/nova-premier-v1", "name": "Nova Premier 1.0", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "nova", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-10-31", "last_updated": "2025-10-31", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 32000}, "cost": {"input": 2.5, "output": 12.5, "cache_read": 0.625}}, "amazon/nova-2-lite-v1": {"id": "amazon/nova-2-lite-v1", "name": "Nova 2 Lite", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "nova", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2025-12-02", "last_updated": "2025-12-02", "modalities": {"input": ["text", "image", "video", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 65535}, "cost": {"input": 0.3, "output": 2.5}}, "amazon/nova-pro-v1": {"id": "amazon/nova-pro-v1", "name": "Nova Pro 1.0", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "nova-pro", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-10-31", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 300000, "output": 5120}, "cost": {"input": 0.8, "output": 3.2}}, "amazon/nova-micro-v1": {"id": "amazon/nova-micro-v1", "name": "Nova Micro 1.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova-micro", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-10-31", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 5120}, "cost": {"input": 0.035, "output": 0.14}}, "amazon/nova-lite-v1": {"id": "amazon/nova-lite-v1", "name": "Nova Lite 1.0", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "nova-lite", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-10-31", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 300000, "output": 5120}, "cost": {"input": 0.06, "output": 0.24}}, "~moonshotai/kimi-latest": {"id": "~moonshotai/kimi-latest", "name": "MoonshotAI Kimi Latest", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "high", "max"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1048576, "output": 1048576}, "cost": {"input": 2.8, "output": 14, "cache_read": 0.29}}, "ibm-granite/granite-4.0-h-micro": {"id": "ibm-granite/granite-4.0-h-micro", "name": "Granite 4.0 Micro", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "granite", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-10-20", "last_updated": "2025-10-20", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131000, "output": 131000}, "cost": {"input": 0.017, "output": 0.112}}, "ibm-granite/granite-4.1-8b": {"id": "ibm-granite/granite-4.1-8b", "name": "Granite 4.1 8B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "family": "granite", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-30", "last_updated": "2026-04-30", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.05, "output": 0.1, "cache_read": 0.05}}, "x-ai/grok-4.20-multi-agent": {"id": "x-ai/grok-4.20-multi-agent", "name": "Grok 4.20 Multi-Agent", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}], "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2025-09-01", "release_date": "2026-03-31", "last_updated": "2026-03-31", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 2000000, "output": 2000000}, "cost": {"input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [{"input": 2.5, "output": 5, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 5, "cache_read": 0.4}}}, "x-ai/grok-4.3": {"id": "x-ai/grok-4.3", "name": "Grok 4.3", "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 1000000}, "cost": {"input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [{"input": 2.5, "output": 5, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 5, "cache_read": 0.4}}}, "x-ai/grok-4.5": {"id": "x-ai/grok-4.5", "name": "Grok 4.5", "description": "xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-08", "last_updated": "2026-07-08", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 500000, "output": 500000}, "cost": {"input": 2, "output": 6, "cache_read": 0.3, "tiers": [{"input": 4, "output": 12, "cache_read": 0.6, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 12, "cache_read": 0.6}}}, "x-ai/grok-4.6": {"id": "x-ai/grok-4.6", "name": "Grok 4.6", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-02-01", "release_date": "2026-08-12", "last_updated": "2026-08-12", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 500000, "output": 500000}, "cost": {"input": 2, "output": 6, "cache_read": 0.5, "tiers": [{"input": 4, "output": 12, "cache_read": 1, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 4, "output": 12, "cache_read": 1}}}, "x-ai/grok-build-0.1": {"id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", "description": "Fast Grok coding model tuned for agentic engineering and iterative edits", "family": "grok-build", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 256000, "output": 256000}, "cost": {"input": 1, "output": 2, "cache_read": 0.2, "tiers": [{"input": 2, "output": 4, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2, "output": 4, "cache_read": 0.4}}}, "x-ai/grok-4.20": {"id": "x-ai/grok-4.20", "name": "Grok 4.20", "description": "Grok model for agentic tool use, reasoning, coding, and live assistance", "family": "grok", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-09-01", "release_date": "2026-03-31", "last_updated": "2026-03-31", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 2000000, "output": 2000000}, "cost": {"input": 1.25, "output": 2.5, "cache_read": 0.2, "tiers": [{"input": 2.5, "output": 5, "cache_read": 0.4, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 2.5, "output": 5, "cache_read": 0.4}}}, "kwaipilot/kat-coder-pro-v2": {"id": "kwaipilot/kat-coder-pro-v2", "name": "KAT-Coder-Pro V2", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "kat-coder", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-27", "last_updated": "2026-03-27", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 80000}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.06}}, "kwaipilot/kat-coder-pro-v2.5": {"id": "kwaipilot/kat-coder-pro-v2.5", "name": "KAT-Coder-Pro V2.5", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "kat-coder", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-10", "last_updated": "2026-07-10", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 256000, "output": 80000}, "cost": {"input": 0.74, "output": 2.96, "cache_read": 0.15}}, "kwaipilot/kat-coder-air-v2.5": {"id": "kwaipilot/kat-coder-air-v2.5", "name": "KAT-Coder-Air V2.5", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "kat-coder", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-10", "last_updated": "2026-07-10", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 256000, "output": 80000}, "cost": {"input": 0.15, "output": 0.6, "cache_read": 0.03}}, "sao10k/l3.1-euryale-70b": {"id": "sao10k/l3.1-euryale-70b", "name": "Llama 3.1 Euryale 70B v2.2", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-08-28", "last_updated": "2024-08-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.85, "output": 0.85}}, "sao10k/l3.3-euryale-70b": {"id": "sao10k/l3.3-euryale-70b", "name": "Llama 3.3 Euryale 70B", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-12-18", "last_updated": "2024-12-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.65, "output": 0.75}}, "sao10k/l3-lunaris-8b": {"id": "sao10k/l3-lunaris-8b", "name": "Llama 3 8B Lunaris", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-08-13", "last_updated": "2024-08-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 8192, "output": 16384}, "cost": {"input": 0.04, "output": 0.05}}, "openrouter/pareto-code": {"id": "openrouter/pareto-code", "name": "Pareto Code Router", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 2000000, "output": 200000}}, "openrouter/bodybuilder": {"id": "openrouter/bodybuilder", "name": "Body Builder (beta)", "description": "Preview model for early access evaluation, prototyping, and compatibility testing", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "release_date": "2025-12-05", "last_updated": "2025-12-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 128000}}, "openrouter/free": {"id": "openrouter/free", "name": "Free Models Router", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "input": 200000, "output": 8000}, "cost": {"input": 0, "output": 0}}, "openrouter/auto": {"id": "openrouter/auto", "name": "Auto Router", "description": "Image model for prompt-driven generation, editing, and visual design workflows", "family": "auto", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2023-11-08", "last_updated": "2023-11-08", "modalities": {"input": ["text", "image", "audio", "pdf", "video"], "output": ["text", "image"]}, "open_weights": false, "limit": {"context": 2000000, "output": 2000000}}, "openrouter/fusion": {"id": "openrouter/fusion", "name": "Fusion", "description": "General-purpose chat model for instruction following, writing, and analysis", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": false, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}}, "aion-labs/aion-2.0": {"id": "aion-labs/aion-2.0", "name": "Aion-2.0", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.8, "output": 1.6, "cache_read": 0.2}}, "aion-labs/aion-3.0-mini": {"id": "aion-labs/aion-3.0-mini", "name": "Aion-3.0-Mini", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-07", "last_updated": "2026-07-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.7, "output": 1.4, "cache_read": 0.18}}, "aion-labs/aion-rp-llama-3.1-8b": {"id": "aion-labs/aion-rp-llama-3.1-8b", "name": "Aion-RP 1.0 (8B)", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2023-12-31", "release_date": "2025-02-04", "last_updated": "2025-02-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 32768, "output": 32768}, "cost": {"input": 0.8, "output": 1.6}}, "aion-labs/aion-3.0": {"id": "aion-labs/aion-3.0", "name": "Aion-3.0", "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-07-07", "last_updated": "2026-07-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 3, "output": 6, "cache_read": 0.75}}, "xiaomi/mimo-v2.5": {"id": "xiaomi/mimo-v2.5", "name": "MiMo-V2.5", "description": "Open MiMo model for multimodal coding agents and long-context automation", "family": "mimo", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": {"input": ["text", "image", "audio", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1050000, "output": 131072}, "cost": {"input": 0.14, "output": 0.28, "cache_read": 0.0028}}, "xiaomi/mimo-v2.5-pro": {"id": "xiaomi/mimo-v2.5-pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1050000, "output": 131072}, "cost": {"input": 0.435, "output": 0.87, "cache_read": 0.0036}}, "writer/palmyra-x5": {"id": "writer/palmyra-x5", "name": "Palmyra X5", "description": "General-purpose chat model for instruction following, writing, and analysis", "family": "palmyra", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2026-01-21", "last_updated": "2026-01-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1040000, "output": 8192}, "cost": {"input": 0.6, "output": 6}}, "anthropic/claude-sonnet-4.6": {"id": "anthropic/claude-sonnet-4.6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "max"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [{"input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5}}}, "anthropic/claude-fable-5": {"id": "anthropic/claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5}}, "anthropic/claude-opus-4.8-fast": {"id": "anthropic/claude-opus-4.8-fast", "name": "Claude Opus 4.8 (Fast)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5}}, "anthropic/claude-opus-4.1": {"id": "anthropic/claude-opus-4.1", "name": "Claude Opus 4.1 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1024, "max": 31999}], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 32000}, "cost": {"input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75}}, "anthropic/claude-opus-4.5": {"id": "anthropic/claude-opus-4.5", "name": "Claude Opus 4.5 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high"]}, {"type": "budget_tokens", "min": 1024, "max": 63999}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 64000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "anthropic/claude-opus-4.7": {"id": "anthropic/claude-opus-4.7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [{"input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5}}}, "anthropic/claude-sonnet-4.5": {"id": "anthropic/claude-sonnet-4.5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1024, "max": 63999}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 64000}, "cost": {"input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [{"input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5}}}, "anthropic/claude-3-haiku": {"id": "anthropic/claude-3-haiku", "name": "Claude 3 Haiku", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-08-31", "release_date": "2024-03-13", "last_updated": "2024-03-13", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 4096}, "cost": {"input": 0.25, "output": 1.25, "cache_read": 0.03, "cache_write": 0.3}}, "anthropic/claude-opus-5-fast": {"id": "anthropic/claude-opus-5-fast", "name": "Claude Opus 5 (Fast)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-05", "release_date": "2026-07-24", "last_updated": "2026-07-24", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5}}, "anthropic/claude-sonnet-4": {"id": "anthropic/claude-sonnet-4", "name": "Claude Sonnet 4", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1024, "max": 63999}], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": {"input": ["image", "text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 64000}, "cost": {"input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75, "tiers": [{"input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 6, "output": 22.5, "cache_read": 0.6, "cache_write": 7.5}}}, "anthropic/claude-haiku-4.5": {"id": "anthropic/claude-haiku-4.5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1024, "max": 63999}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 64000}, "cost": {"input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25}}, "anthropic/claude-opus-4": {"id": "anthropic/claude-opus-4", "name": "Claude Opus 4", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1024, "max": 31999}], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-01-31", "release_date": "2025-05-22", "last_updated": "2025-05-22", "modalities": {"input": ["image", "text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 32000}, "cost": {"input": 15, "output": 75, "cache_read": 1.5, "cache_write": 18.75}}, "anthropic/claude-opus-4.8": {"id": "anthropic/claude-opus-4.8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "anthropic/claude-sonnet-5": {"id": "anthropic/claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-30", "last_updated": "2026-06-30", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5}}, "anthropic/claude-opus-5": {"id": "anthropic/claude-opus-5", "name": "Claude Opus 5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-05", "release_date": "2026-07-24", "last_updated": "2026-07-24", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "anthropic/claude-opus-4.6": {"id": "anthropic/claude-opus-4.6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "max"]}, {"type": "budget_tokens"}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-05", "last_updated": "2026-03-13", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25, "tiers": [{"input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5, "tier": {"type": "context", "size": 200000}}], "context_over_200k": {"input": 10, "output": 37.5, "cache_read": 1, "cache_write": 12.5}}}, "anthropic/claude-opus-4.7-fast": {"id": "anthropic/claude-opus-4.7-fast", "name": "Claude Opus 4.7 (Fast)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-16", "last_updated": "2026-04-16", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 30, "output": 150, "cache_read": 3, "cache_write": 37.5}}, "liquid/lfm-2.5-2.6b:free": {"id": "liquid/lfm-2.5-2.6b:free", "name": "LFM2.5-2.6B (free)", "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads", "family": "liquid", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-08-11", "last_updated": "2026-08-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 32768}, "cost": {"input": 0, "output": 0}}, "~anthropic/claude-sonnet-latest": {"id": "~anthropic/claude-sonnet-latest", "name": "Anthropic Claude Sonnet Latest", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5}}, "~anthropic/claude-opus-latest": {"id": "~anthropic/claude-opus-latest", "name": "Claude Opus Latest", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "~anthropic/claude-haiku-latest": {"id": "~anthropic/claude-haiku-latest", "name": "Anthropic Claude Haiku Latest", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "budget_tokens", "min": 1024, "max": 63999}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-27", "last_updated": "2026-04-27", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 64000}, "cost": {"input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25}}, "~anthropic/claude-fable-latest": {"id": "~anthropic/claude-fable-latest", "name": "Claude Fable Latest", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-06-09", "last_updated": "2026-06-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5}}, "z-ai/glm-4.6v": {"id": "z-ai/glm-4.6v", "name": "GLM-4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.3, "output": 0.9, "cache_read": 0.055}}, "z-ai/glm-5": {"id": "z-ai/glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.95, "output": 2.55, "cache_read": 0.2}}, "z-ai/glm-4.5-air": {"id": "z-ai/glm-4.5-air", "name": "GLM-4.5-Air", "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 98304}, "cost": {"input": 0.13, "output": 0.85, "cache_read": 0.025}}, "z-ai/glm-5.1": {"id": "z-ai/glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 1.4, "output": 4.4, "cache_read": 0.26}}, "z-ai/glm-4.7-flash": {"id": "z-ai/glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Budget GLM lane for fast coding help, routing, and everyday automation", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 202752, "output": 16384}, "cost": {"input": 0.06, "output": 0.4, "cache_read": 0.01}}, "z-ai/glm-5.2": {"id": "z-ai/glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "xhigh"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 131072}, "cost": {"input": 0.5, "output": 3.15, "cache_read": 0.1}}, "z-ai/glm-4.6": {"id": "z-ai/glm-4.6", "name": "GLM-4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.5, "output": 2, "cache_read": 0.1}}, "z-ai/glm-4.5": {"id": "z-ai/glm-4.5", "name": "GLM-4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 98304}, "cost": {"input": 0.6, "output": 2.2, "cache_read": 0.11}}, "z-ai/glm-4.5v": {"id": "z-ai/glm-4.5v", "name": "GLM-4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 65536, "output": 16384}, "cost": {"input": 0.6, "output": 1.8, "cache_read": 0.11}}, "z-ai/glm-4.7": {"id": "z-ai/glm-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.4, "output": 1.75, "cache_read": 0.08}}, "z-ai/glm-5-turbo": {"id": "z-ai/glm-5-turbo", "name": "GLM-5-Turbo", "description": "Faster GLM-5 lane for coding agents that need lower latency", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": false, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 202752, "output": 131072}, "cost": {"input": 1.2, "output": 4, "cache_read": 0.24}}, "z-ai/glm-5v-turbo": {"id": "z-ai/glm-5v-turbo", "name": "GLM-5V-Turbo", "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": {"input": ["image", "text", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 202752, "output": 131072}, "cost": {"input": 1.2, "output": 4, "cache_read": 0.24}}, "perplexity/sonar-pro-search": {"id": "perplexity/sonar-pro-search", "name": "Sonar Pro Search", "description": "Advanced Sonar search model for deeper research and cited synthesis", "family": "sonar-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": true, "release_date": "2025-10-30", "last_updated": "2025-10-30", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 8000}, "cost": {"input": 3, "output": 15}}, "perplexity/sonar-deep-research": {"id": "perplexity/sonar-deep-research", "name": "Sonar Deep Research", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "family": "sonar-deep-research", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-03-07", "last_updated": "2025-03-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 2, "output": 8, "reasoning": 3}}, "perplexity/sonar": {"id": "perplexity/sonar", "name": "Sonar", "description": "Sonar search model for current answers, retrieval, and citation-backed chat", "family": "sonar", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-01-27", "last_updated": "2025-01-27", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 127072, "output": 127072}, "cost": {"input": 1, "output": 1}}, "perplexity/sonar-pro": {"id": "perplexity/sonar-pro", "name": "Sonar Pro", "description": "Advanced Sonar search model for deeper research and cited synthesis", "family": "sonar-pro", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-03-07", "last_updated": "2025-03-07", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 8000}, "cost": {"input": 3, "output": 15}}, "perplexity/sonar-reasoning-pro": {"id": "perplexity/sonar-reasoning-pro", "name": "Sonar Reasoning Pro", "description": "Web-grounded reasoning model for multi-step research and cited answers", "family": "sonar-reasoning", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "release_date": "2025-03-07", "last_updated": "2025-03-07", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 128000}, "cost": {"input": 2, "output": 8}}, "moonshotai/kimi-k2.5": {"id": "moonshotai/kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.57, "output": 2.85, "cache_read": 0.095}}, "moonshotai/kimi-k2.6": {"id": "moonshotai/kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.95, "output": 4, "cache_read": 0.16}}, "moonshotai/kimi-k2.7-code": {"id": "moonshotai/kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.67, "output": 3.4, "cache_read": 0.15}}, "moonshotai/kimi-k2": {"id": "moonshotai/kimi-k2", "name": "Kimi K2 0711", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2024-12-31", "release_date": "2025-07-11", "last_updated": "2025-07-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 100352}, "cost": {"input": 0.57, "output": 2.3}}, "moonshotai/kimi-k2-thinking": {"id": "moonshotai/kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "structured_output": true, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 100352}, "cost": {"input": 0.6, "output": 2.5, "cache_read": 0.15}}, "moonshotai/kimi-k3": {"id": "moonshotai/kimi-k3", "name": "Kimi K3", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k3", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "high", "max"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-16", "last_updated": "2026-07-16", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 1048576}, "cost": {"input": 3, "output": 15, "cache_read": 0.3}}, "moonshotai/kimi-k2-0905": {"id": "moonshotai/kimi-k2-0905", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-12-31", "release_date": "2025-09-04", "last_updated": "2025-09-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 100352}, "cost": {"input": 0.6, "output": 2.5}}, "openai/gpt-5.1-codex-mini": {"id": "openai/gpt-5.1-codex-mini", "name": "GPT-5.1 Codex mini", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 0.25, "output": 2, "cache_read": 0.03}}, "openai/gpt-chat-latest": {"id": "openai/gpt-chat-latest", "name": "GPT Chat Latest", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-05-05", "last_updated": "2026-05-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "output": 128000}, "cost": {"input": 5, "output": 30, "cache_read": 0.5}}, "openai/gpt-5.2-pro": {"id": "openai/gpt-5.2-pro", "name": "GPT-5.2 Pro", "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": {"input": ["image", "text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 21, "output": 168}}, "openai/gpt-5.5-pro": {"id": "openai/gpt-5.5-pro", "name": "GPT-5.5 Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 30, "output": 180, "tiers": [{"input": 60, "output": 270, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 60, "output": 270}}}, "openai/gpt-4.1-mini": {"id": "openai/gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1047576, "output": 32768}, "cost": {"input": 0.4, "output": 1.6, "cache_read": 0.1}}, "openai/gpt-4o": {"id": "openai/gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 2.5, "output": 10, "cache_read": 1.25}}, "openai/gpt-5.4-pro": {"id": "openai/gpt-5.4-pro", "name": "GPT-5.4 Pro", "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 30, "output": 180, "tiers": [{"input": 60, "output": 270, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 60, "output": 270}}}, "openai/gpt-audio-mini": {"id": "openai/gpt-audio-mini", "name": "GPT Audio Mini", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "o-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": {"input": ["text", "audio"], "output": ["text", "audio"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 0.6, "output": 2.4}}, "openai/gpt-5.6-sol": {"id": "openai/gpt-5.6-sol", "name": "GPT-5.6 Sol", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt-sol", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25, "tiers": [{"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5}}}, "openai/o3-mini-high": {"id": "openai/o3-mini-high", "name": "o3 Mini High", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-10-31", "release_date": "2025-02-12", "last_updated": "2025-02-12", "modalities": {"input": ["text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "cost": {"input": 1.1, "output": 4.4, "cache_read": 0.55}}, "openai/gpt-4o-mini-2024-07-18": {"id": "openai/gpt-4o-mini-2024-07-18", "name": "GPT-4o-mini (2024-07-18)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "o-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-10-31", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 0.15, "output": 0.6, "cache_read": 0.075}}, "openai/gpt-4.1-nano": {"id": "openai/gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": {"input": ["image", "text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1047576, "output": 32768}, "cost": {"input": 0.1, "output": 0.4, "cache_read": 0.025}}, "openai/gpt-oss-20b": {"id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.03, "output": 0.13, "cache_read": 0.03}}, "openai/gpt-oss-safeguard-20b": {"id": "openai/gpt-oss-safeguard-20b", "name": "gpt-oss-safeguard-20b", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-10-29", "last_updated": "2025-10-29", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 0.075, "output": 0.3, "cache_read": 0.0375}}, "openai/o3-mini": {"id": "openai/o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": {"input": ["text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "cost": {"input": 1.1, "output": 4.4, "cache_read": 0.55}}, "openai/gpt-5.6-sol-pro": {"id": "openai/gpt-5.6-sol-pro", "name": "GPT-5.6 Sol Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-sol", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25, "tiers": [{"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5}}}, "openai/gpt-5.5": {"id": "openai/gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 5, "output": 30, "cache_read": 0.5, "tiers": [{"input": 10, "output": 45, "cache_read": 1, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 10, "output": 45, "cache_read": 1}}}, "openai/gpt-3.5-turbo-0613": {"id": "openai/gpt-3.5-turbo-0613", "name": "GPT-3.5 Turbo (older v0613)", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2021-09-30", "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 4095, "output": 4096}, "cost": {"input": 1, "output": 2}}, "openai/o3-pro": {"id": "openai/o3-pro", "name": "o3-pro", "description": "High-effort o3 tier for difficult technical reasoning and careful answers", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-06-10", "last_updated": "2025-06-10", "modalities": {"input": ["text", "pdf", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "cost": {"input": 20, "output": 80}}, "openai/gpt-4o-mini": {"id": "openai/gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 0.15, "output": 0.6, "cache_read": 0.075}}, "openai/gpt-5": {"id": "openai/gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.25, "output": 10, "cache_read": 0.125}}, "openai/gpt-audio": {"id": "openai/gpt-audio", "name": "GPT Audio", "description": "Speech generation model for controllable voice, narration, and audio delivery", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": {"input": ["text", "audio"], "output": ["text", "audio"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 2.5, "output": 10}}, "openai/o4-mini-high": {"id": "openai/o4-mini-high", "name": "o4 Mini High", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-06-30", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": {"input": ["image", "text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "cost": {"input": 1.1, "output": 4.4, "cache_read": 0.275}}, "openai/gpt-3.5-turbo": {"id": "openai/gpt-3.5-turbo", "name": "GPT-3.5-turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2021-09-01", "release_date": "2023-03-01", "last_updated": "2023-11-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 16385, "output": 4096}, "cost": {"input": 0.5, "output": 1.5}}, "openai/gpt-4o-2024-05-13": {"id": "openai/gpt-4o-2024-05-13", "name": "GPT-4o (2024-05-13)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 5, "output": 15}}, "openai/gpt-4o-2024-11-20": {"id": "openai/gpt-4o-2024-11-20", "name": "GPT-4o (2024-11-20)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-11-20", "last_updated": "2024-11-20", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 2.5, "output": 10, "cache_read": 1.25}}, "openai/gpt-5.4": {"id": "openai/gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [{"input": 5, "output": 22.5, "cache_read": 0.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 5, "output": 22.5, "cache_read": 0.5}}}, "openai/gpt-5.6-luna-pro": {"id": "openai/gpt-5.6-luna-pro", "name": "GPT-5.6 Luna Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-luna", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 0.1, "output": 0.6, "cache_read": 0.01, "cache_write": 0.125, "tiers": [{"input": 0.2, "output": 0.9, "cache_read": 0.02, "cache_write": 0.25, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 0.2, "output": 0.9, "cache_read": 0.02, "cache_write": 0.25}}}, "openai/gpt-5.2-chat": {"id": "openai/gpt-5.2-chat", "name": "GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-10", "last_updated": "2025-12-10", "modalities": {"input": ["pdf", "image", "text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 32000}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "openai/gpt-5.2-codex": {"id": "openai/gpt-5.2-codex", "name": "GPT-5.2 Codex", "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "openai/gpt-5.4-nano": {"id": "openai/gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": {"input": ["pdf", "image", "text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 0.2, "output": 1.25, "cache_read": 0.02}}, "openai/gpt-5-pro": {"id": "openai/gpt-5-pro", "name": "GPT-5 Pro", "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": {"input": ["image", "text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 15, "output": 120}}, "openai/gpt-5.4-mini": {"id": "openai/gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": {"input": ["pdf", "image", "text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 0.75, "output": 4.5, "cache_read": 0.075}}, "openai/gpt-oss-120b": {"id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.03, "output": 0.17, "cache_read": 0.03}}, "openai/gpt-3.5-turbo-16k": {"id": "openai/gpt-3.5-turbo-16k", "name": "GPT-3.5 Turbo 16k", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2021-09-30", "release_date": "2023-08-28", "last_updated": "2023-08-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 16385, "output": 4096}, "cost": {"input": 3, "output": 4}}, "openai/o1": {"id": "openai/o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "cost": {"input": 15, "output": 60, "cache_read": 7.5}}, "openai/gpt-5.6-luna": {"id": "openai/gpt-5.6-luna", "name": "GPT-5.6 Luna", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt-luna", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 0.1, "output": 0.6, "cache_read": 0.01, "cache_write": 0.125, "tiers": [{"input": 0.2, "output": 0.9, "cache_read": 0.02, "cache_write": 0.25, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 0.2, "output": 0.9, "cache_read": 0.02, "cache_write": 0.25}}}, "openai/gpt-5.2": {"id": "openai/gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": {"input": ["pdf", "image", "text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "openai/gpt-5.3-codex": {"id": "openai/gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "openai/gpt-5-mini": {"id": "openai/gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 0.25, "output": 2, "cache_read": 0.025}}, "openai/o1-pro": {"id": "openai/o1-pro", "name": "o1-pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2025-03-19", "last_updated": "2025-03-19", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "cost": {"input": 150, "output": 600}}, "openai/gpt-5-image": {"name": "OpenAI: GPT-5 Image", "release_date": "2025-10-14", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 10.0, "output": 10.0}, "id": "openai/gpt-5-image"}, "openai/gpt-5.1": {"id": "openai/gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": {"input": ["image", "text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.25, "output": 10, "cache_read": 0.125}}, "openai/gpt-4-turbo-preview": {"id": "openai/gpt-4-turbo-preview", "name": "GPT-4 Turbo Preview", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 10, "output": 30}}, "openai/gpt-4-turbo": {"id": "openai/gpt-4-turbo", "name": "GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 4096}, "cost": {"input": 10, "output": 30}}, "openai/gpt-4o-2024-08-06": {"id": "openai/gpt-4o-2024-08-06", "name": "GPT-4o (2024-08-06)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-08-06", "last_updated": "2024-08-06", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 2.5, "output": 10, "cache_read": 1.25}}, "openai/gpt-5.4-image-2": {"name": "OpenAI: GPT-5.4 Image 2", "release_date": "2026-04-22", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 8.0, "output": 15.0}, "id": "openai/gpt-5.4-image-2"}, "openai/gpt-5.1-codex": {"id": "openai/gpt-5.1-codex", "name": "GPT-5.1 Codex", "description": "Codex GPT for repository edits, code review, and practical software agents", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.25, "output": 10, "cache_read": 0.13}}, "openai/gpt-5-nano": {"id": "openai/gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 0.05, "output": 0.4, "cache_read": 0.005}}, "openai/o3": {"id": "openai/o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "cost": {"input": 2, "output": 8, "cache_read": 0.5}}, "openai/gpt-3.5-turbo-instruct": {"id": "openai/gpt-3.5-turbo-instruct", "name": "GPT-3.5 Turbo Instruct", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2021-09-30", "release_date": "2023-09-28", "last_updated": "2023-09-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 4095, "output": 4096}, "cost": {"input": 1.5, "output": 2}}, "openai/gpt-5.6-terra": {"id": "openai/gpt-5.6-terra", "name": "GPT-5.6 Terra", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt-terra", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 1, "output": 6, "cache_read": 0.1, "cache_write": 1.25, "tiers": [{"input": 2, "output": 9, "cache_read": 0.2, "cache_write": 2.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 2, "output": 9, "cache_read": 0.2, "cache_write": 2.5}}}, "openai/gpt-5-image-mini": {"name": "OpenAI: GPT-5 Image Mini", "release_date": "2025-10-16", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 2.5, "output": 2}, "id": "openai/gpt-5-image-mini"}, "openai/gpt-5.6-terra-pro": {"id": "openai/gpt-5.6-terra-pro", "name": "GPT-5.6 Terra Pro", "description": "Frontier GPT model for professional reasoning, coding, and multimodal work", "family": "gpt-terra", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 1, "output": 6, "cache_read": 0.1, "cache_write": 1.25, "tiers": [{"input": 2, "output": 9, "cache_read": 0.2, "cache_write": 2.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 2, "output": 9, "cache_read": 0.2, "cache_write": 2.5}}}, "openai/gpt-4.1": {"id": "openai/gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1047576, "output": 32768}, "cost": {"input": 2, "output": 8, "cache_read": 0.5}}, "openai/o4-mini": {"id": "openai/o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": {"input": ["image", "text", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "cost": {"input": 1.1, "output": 4.4, "cache_read": 0.275}}, "openai/gpt-oss-20b:free": {"id": "openai/gpt-oss-20b:free", "name": "gpt-oss-20b (free)", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0, "output": 0}}, "openai/gpt-5.1-codex-max": {"id": "openai/gpt-5.1-codex-max", "name": "GPT-5.1 Codex Max", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.25, "output": 10, "cache_read": 0.125}}, "openai/gpt-4": {"id": "openai/gpt-4", "name": "GPT-4", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-11", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 8191, "output": 4096}, "cost": {"input": 30, "output": 60}}, "baidu/ernie-4.5-vl-424b-a47b": {"id": "baidu/ernie-4.5-vl-424b-a47b", "name": "ERNIE 4.5 VL 424B A47B ", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "ernie", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-06-30", "last_updated": "2025-06-30", "modalities": {"input": ["image", "text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 123000, "output": 16000}, "cost": {"input": 0.42, "output": 1.25}}, "meta-llama/llama-3.1-70b-instruct": {"id": "meta-llama/llama-3.1-70b-instruct", "name": "Llama 3.1 70B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.4, "output": 0.4}}, "meta-llama/llama-guard-4-12b": {"id": "meta-llama/llama-guard-4-12b", "name": "Llama Guard 4 12B", "description": "Safety model for policy screening, moderation, and risk-aware routing workflows", "family": "llama", "attachment": true, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-04-30", "last_updated": "2025-04-30", "modalities": {"input": ["image", "text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 16384}, "cost": {"input": 0.18, "output": 0.18}}, "meta-llama/llama-3.2-1b-instruct": {"id": "meta-llama/llama-3.2-1b-instruct", "name": "Llama 3.2 1B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 60000, "output": 60000}, "cost": {"input": 0.027, "output": 0.201}}, "meta-llama/llama-3.3-70b-instruct": {"id": "meta-llama/llama-3.3-70b-instruct", "name": "Llama-3.3-70B-Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.1, "output": 0.32}}, "meta-llama/llama-4-maverick": {"id": "meta-llama/llama-4-maverick", "name": "Llama 4 Maverick", "description": "Open multimodal Llama model for strong reasoning and fast responses", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 16384}, "cost": {"input": 0.2, "output": 0.696}}, "meta-llama/llama-4-scout": {"id": "meta-llama/llama-4-scout", "name": "Llama 4 Scout", "description": "Open multimodal Llama model for long-context analysis and efficient agents", "family": "llama", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-08-31", "release_date": "2025-04-05", "last_updated": "2025-04-05", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1310720, "output": 16384}, "cost": {"input": 0.1, "output": 0.3}}, "meta-llama/llama-3.2-3b-instruct": {"id": "meta-llama/llama-3.2-3b-instruct", "name": "Llama 3.2 3B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-09-25", "last_updated": "2024-09-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.05, "output": 0.33}}, "meta-llama/llama-3.1-8b-instruct": {"id": "meta-llama/llama-3.1-8b-instruct", "name": "Llama 3.1 8B Instruct", "description": "Open Llama instruction model for multilingual chat, reasoning, and coding", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12-31", "release_date": "2024-07-23", "last_updated": "2024-07-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.05, "output": 0.08, "cache_read": 0.025}}, "arcee-ai/virtuoso-large": {"id": "arcee-ai/virtuoso-large", "name": "Virtuoso Large", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-03-31", "release_date": "2025-05-05", "last_updated": "2025-05-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 131072, "output": 64000}, "cost": {"input": 0.75, "output": 1.2}}, "arcee-ai/trinity-large-thinking": {"id": "arcee-ai/trinity-large-thinking", "name": "Trinity Large Thinking", "description": "Flagship model for demanding analysis, coding, and production agent workflows", "family": "trinity", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.22, "output": 0.85, "cache_read": 0.06}}, "bytedance-seed/seed-1.6": {"id": "bytedance-seed/seed-1.6", "name": "Seed 1.6", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": {"input": ["image", "text", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.25, "output": 2, "tiers": [{"input": 0.5, "output": 4, "tier": {"type": "context", "size": 128000}}]}}, "bytedance-seed/seed-2.0-mini": {"id": "bytedance-seed/seed-2.0-mini", "name": "Seed-2.0-Mini", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-26", "last_updated": "2026-02-26", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 131072}, "cost": {"input": 0.1, "output": 0.4, "tiers": [{"input": 0.2, "output": 0.8, "tier": {"type": "context", "size": 128000}}]}}, "bytedance-seed/seed-1.6-flash": {"id": "bytedance-seed/seed-1.6-flash", "name": "Seed 1.6 Flash", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": {"input": ["image", "text", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.075, "output": 0.3, "tiers": [{"input": 0.1, "output": 0.8, "tier": {"type": "context", "size": 128000}}]}}, "bytedance-seed/seed-2-1-turbo": {"id": "bytedance-seed/seed-2-1-turbo", "name": "Seed 2.1 Turbo", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-08-12", "last_updated": "2026-08-12", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.5, "output": 2.5}}, "bytedance-seed/seed-2.0-lite": {"id": "bytedance-seed/seed-2.0-lite", "name": "Seed-2.0-Lite", "description": "Multimodal reasoning model for visual analysis, planning, and tool use", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-03-10", "last_updated": "2026-03-10", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 131072}, "cost": {"input": 0.25, "output": 2, "tiers": [{"input": 0.5, "output": 4, "tier": {"type": "context", "size": 128000}}]}}, "bytedance-seed/seed-2.0-code": {"id": "bytedance-seed/seed-2.0-code", "name": "Seed 2.0 Code", "description": "Coding model for repository understanding, refactors, and agentic engineering tasks", "family": "seed", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-14", "last_updated": "2026-02-14", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": false, "limit": {"context": 262144, "output": 131072}, "cost": {"input": 0.5, "output": 3, "tiers": [{"input": 1, "output": 6, "tier": {"type": "context", "size": 128000}}]}}, "stepfun/step-3.5-flash": {"id": "stepfun/step-3.5-flash", "name": "Step 3.5 Flash", "description": "StepFun flash lane for quick multimodal reasoning and coding assistance", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-29", "last_updated": "2026-02-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.1, "output": 0.3}}, "stepfun/step-3.7-flash": {"id": "stepfun/step-3.7-flash", "name": "Step 3.7 Flash", "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2026-03-01", "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "input": 256000, "output": 256000}, "cost": {"input": 0.2, "output": 1.15, "cache_read": 0.04}}, "anthracite-org/magnum-v4-72b": {"id": "anthracite-org/magnum-v4-72b", "name": "Magnum v4 72B", "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": true, "temperature": true, "knowledge": "2024-06-30", "release_date": "2024-10-22", "last_updated": "2024-10-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32768, "output": 4096}, "cost": {"input": 3, "output": 5}}, "microsoft/mai-image-2.5-pro": {"name": "MAI-Image-2.5 Pro", "release_date": "2026-07-24", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0.0, "output": 5.0}, "id": "microsoft/mai-image-2.5-pro"}, "microsoft/mai-voice-2-flash": {"name": "MAI-Voice-2-Flash", "release_date": "2026-07-23", "modalities": {"input": ["text"], "output": ["speech"]}, "cost": {"input": 0, "output": 22}, "voices": ["en-US-Harper:MAI-Voice-2", "es-MX-Valeria:MAI-Voice-2", "fr-FR-Soleil:MAI-Voice-2", "de-DE-Klaus:MAI-Voice-2"], "defaults": {"voice": "en-US-Harper:MAI-Voice-2"}, "id": "microsoft/mai-voice-2-flash"}, "microsoft/mai-voice-2": {"name": "MAI-Voice-2", "release_date": "2026-06-03", "modalities": {"input": ["text"], "output": ["speech"]}, "cost": {"input": 0, "output": 22}, "voices": ["en-US-Harper:MAI-Voice-2", "es-MX-Valeria:MAI-Voice-2", "fr-FR-Soleil:MAI-Voice-2", "de-DE-Klaus:MAI-Voice-2"], "defaults": {"voice": "en-US-Harper:MAI-Voice-2"}, "id": "microsoft/mai-voice-2"}, "x-ai/grok-voice-tts-1.0": {"name": "Grok Voice TTS 1.0", "release_date": "2026-05-15", "modalities": {"input": ["text"], "output": ["speech"]}, "cost": {"input": 0, "output": 15}, "voices": ["eve", "ara", "rex", "sal", "leo"], "defaults": {"voice": "eve"}, "id": "x-ai/grok-voice-tts-1.0"}, "google/gemini-3.1-flash-tts-preview": {"name": "Gemini 3.1 Flash TTS", "release_date": "2026-04-24", "modalities": {"input": ["text"], "output": ["speech"]}, "cost": {"input": 1, "output": 22}, "voices": ["Zephyr", "Puck", "Charon", "Kore", "Fenrir", "Leda", "Orus", "Aoede", "Callirrhoe", "Autonoe", "Enceladus", "Iapetus", "Umbriel", "Algieba", "Despina", "Erinome", "Algenib", "Rasalgethi", "Laomedeia", "Achernar", "Alnilam", "Schedar", "Gacrux", "Pulcherrima", "Achird", "Zubenelgenubi", "Vindemiatrix", "Sadachbia", "Sadaltager", "Sulafat"], "defaults": {"voice": "Kore", "response_format": "pcm"}, "id": "google/gemini-3.1-flash-tts-preview"}, "mistralai/voxtral-mini-tts-2603": {"name": "Voxtral Mini TTS", "release_date": "2026-03-26", "modalities": {"input": ["text"], "output": ["speech"]}, "cost": {"input": 0.05, "output": 0.2}, "voices": ["en_paul_sad", "en_paul_neutral", "en_paul_happy", "en_paul_frustrated", "en_paul_excited", "en_paul_confident", "en_paul_cheerful", "en_paul_angry", "gb_oliver_neutral", "gb_oliver_sad", "gb_oliver_excited", "gb_oliver_curious", "gb_oliver_confident", "gb_oliver_cheerful", "gb_oliver_angry", "gb_jane_sarcasm", "gb_jane_confused", "gb_jane_shameful", "gb_jane_sad", "gb_jane_neutral", "gb_jane_jealousy", "gb_jane_frustrated", "gb_jane_curious", "gb_jane_confident", "fr_marie_sad", "fr_marie_neutral", "fr_marie_happy", "fr_marie_excited", "fr_marie_curious", "fr_marie_angry"], "defaults": {"voice": "en_paul_cheerful"}, "id": "mistralai/voxtral-mini-tts-2603"}, "recraft/recraft-v4.1-pro-vector": {"name": "Recraft V4.1 Pro Vector", "release_date": "2026-05-14", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.3, "type": "request"}, "id": "recraft/recraft-v4.1-pro-vector"}, "recraft/recraft-v4.1-vector": {"name": "Recraft V4.1 Vector", "release_date": "2026-05-14", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.08, "type": "request"}, "id": "recraft/recraft-v4.1-vector"}, "recraft/recraft-v4.1-pro": {"name": "Recraft V4.1 Pro", "release_date": "2026-05-14", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.25, "type": "request"}, "id": "recraft/recraft-v4.1-pro"}, "recraft/recraft-v4.1": {"name": "Recraft V4.1", "release_date": "2026-05-14", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.04, "type": "request"}, "id": "recraft/recraft-v4.1"}, "recraft/recraft-v3": {"name": "Recraft V3", "release_date": "2026-05-06", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.04, "type": "request"}, "id": "recraft/recraft-v3"}, "sourceful/riverflow-v2.5-pro": {"name": "Sourceful: Riverflow V2.5 Pro", "release_date": "2026-06-04", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.24, "type": "request"}, "id": "sourceful/riverflow-v2.5-pro"}, "sourceful/riverflow-v2.5-fast": {"name": "Sourceful: Riverflow V2 Fast", "release_date": "2026-06-04", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.02, "type": "request"}, "id": "sourceful/riverflow-v2.5-fast"}, "sourceful/riverflow-v2-standard-preview": {"name": "Sourceful: Riverflow V2 Standard Preview", "release_date": "2025-12-09", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 8.38}, "id": "sourceful/riverflow-v2-standard-preview"}, "sourceful/riverflow-v2-max-preview": {"name": "Sourceful: Riverflow V2 Max Preview", "release_date": "2025-12-09", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 17.96}, "id": "sourceful/riverflow-v2-max-preview"}, "black-forest-labs/flux.2-max": {"name": "Black Forest Labs: FLUX.2 Max", "release_date": "2025-12-16", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 7.32, "output": 7.32}, "id": "black-forest-labs/flux.2-max"}, "black-forest-labs/flux.2-pro": {"name": "Black Forest Labs: FLUX.2 Pro", "release_date": "2025-11-25", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 3.66, "output": 3.66}, "id": "black-forest-labs/flux.2-pro"}, "black-forest-labs/flux.2-flex": {"name": "Black Forest Labs: FLUX.2 Flex", "release_date": "2025-11-25", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 14.64, "output": 14.64}, "id": "black-forest-labs/flux.2-flex"}}}, "huggingface": {"id": "huggingface", "env": ["HF_TOKEN"], "npm": "@ai-sdk/openai-compatible", "api": "https://router.huggingface.co/v1", "name": "Hugging Face", "doc": "https://huggingface.co/docs/inference-providers", "models": {"google/gemma-4-26B-A4B-it": {"id": "google/gemma-4-26B-A4B-it", "name": "Gemma 4 26B A4B IT", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.13, "output": 0.4}}, "google/gemma-4-31B-it": {"id": "google/gemma-4-31B-it", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.14, "output": 0.4}}, "thinkingmachines/Inkling-Small": {"id": "thinkingmachines/Inkling-Small", "name": "Inkling Small", "description": "Efficient model for low-latency assistance, extraction, and routine automation", "family": "ling", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2026-07-30", "last_updated": "2026-07-30", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 524288, "output": 1048576}, "cost": {"input": 0.5, "output": 1.2}}, "thinkingmachines/Inkling": {"id": "thinkingmachines/Inkling", "name": "Inkling", "description": "Multimodal model for analyzing text, images, documents, and rich media", "family": "ling", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-15", "last_updated": "2026-07-15", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 1048576}, "cost": {"input": 1, "output": 4.05}}, "zai-org/GLM-5": {"id": "zai-org/GLM-5", "name": "GLM-5", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-02-11", "last_updated": "2026-02-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 202752, "output": 131072}, "cost": {"input": 1, "output": 3.2, "cache_read": 0.2}}, "zai-org/GLM-4.5": {"id": "zai-org/GLM-4.5", "name": "GLM-4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 98304}, "cost": {"input": 0.6, "output": 2.2}}, "zai-org/GLM-4.5-Air": {"id": "zai-org/GLM-4.5-Air", "name": "GLM-4.5-Air", "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 98304}, "cost": {"input": 0.13, "output": 0.85}}, "zai-org/GLM-4.5V": {"id": "zai-org/GLM-4.5V", "name": "GLM-4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 65536, "output": 16384}, "cost": {"input": 0.6, "output": 1.8}}, "zai-org/GLM-4.7-Flash": {"id": "zai-org/GLM-4.7-Flash", "name": "GLM-4.7-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-08", "last_updated": "2025-08-08", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 200000, "output": 128000}, "cost": {"input": 0, "output": 0}}, "zai-org/GLM-5.2": {"id": "zai-org/GLM-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 131072}, "cost": {"input": 1.4, "output": 4.4}}, "zai-org/GLM-5.1": {"id": "zai-org/GLM-5.1", "name": "GLM-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-04-03", "last_updated": "2026-04-03", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 202752, "output": 131072}, "cost": {"input": 1, "output": 3.2, "cache_read": 0.2}}, "zai-org/GLM-4.6": {"id": "zai-org/GLM-4.6", "name": "GLM-4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.55, "output": 2.2}}, "zai-org/GLM-4.7": {"id": "zai-org/GLM-4.7", "name": "GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.6, "output": 2.2, "cache_read": 0.11}}, "tencent/Hy3": {"id": "tencent/Hy3", "name": "Hy3", "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks", "family": "Hy", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-06", "last_updated": "2026-07-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 64000}, "cost": {"input": 0.14, "output": 0.58}}, "Qwen/Qwen3.5-27B": {"id": "Qwen/Qwen3.5-27B", "name": "Qwen3.5 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.3, "output": 2.4}}, "Qwen/Qwen3.5-9B": {"id": "Qwen/Qwen3.5-9B", "name": "Qwen3.5 9B", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.17, "output": 0.25}}, "Qwen/Qwen3-235B-A22B-Instruct-2507": {"id": "Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "Qwen3 235B-A22B Instruct 2507", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-07-21", "last_updated": "2025-07-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 16384}, "cost": {"input": 0.855, "output": 2.565}}, "Qwen/Qwen3.5-122B-A10B": {"id": "Qwen/Qwen3.5-122B-A10B", "name": "Qwen3.5 122B-A10B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.4, "output": 3.2}}, "Qwen/Qwen3-Coder-30B-A3B-Instruct": {"id": "Qwen/Qwen3-Coder-30B-A3B-Instruct", "name": "Qwen3-Coder 30B-A3B Instruct", "description": "Smaller Qwen coder for efficient local agents and repo-level fixes", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.07, "output": 0.26}}, "Qwen/Qwen3-235B-A22B": {"id": "Qwen/Qwen3-235B-A22B", "name": "Qwen3 235B-A22B", "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 40960, "output": 16384}, "cost": {"input": 0.2, "output": 0.8}}, "Qwen/Qwen3-Embedding-4B": {"id": "Qwen/Qwen3-Embedding-4B", "name": "Qwen 3 Embedding 4B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32000, "output": 2048}, "cost": {"input": 0.01, "output": 0}}, "Qwen/Qwen3.5-35B-A3B": {"id": "Qwen/Qwen3.5-35B-A3B", "name": "Qwen3.5 35B-A3B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-02-23", "last_updated": "2026-02-23", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.25, "output": 2}}, "Qwen/Qwen3-Coder-Next": {"id": "Qwen/Qwen3-Coder-Next", "name": "Qwen3-Coder-Next", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-03", "last_updated": "2026-02-03", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.2, "output": 1.5}}, "Qwen/Qwen3.6-27B": {"id": "Qwen/Qwen3.6-27B", "name": "Qwen3.6 27B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.47, "output": 3.19}}, "Qwen/Qwen3.5-397B-A17B": {"id": "Qwen/Qwen3.5-397B-A17B", "name": "Qwen3.5-397B-A17B", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2025-04", "release_date": "2026-02-01", "last_updated": "2026-02-01", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 32768}, "cost": {"input": 0.6, "output": 3.6}}, "Qwen/Qwen3-Embedding-8B": {"id": "Qwen/Qwen3-Embedding-8B", "name": "Qwen 3 Embedding 8B", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-12", "release_date": "2025-01-01", "last_updated": "2025-01-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 32000, "output": 4096}, "cost": {"input": 0.01, "output": 0}}, "Qwen/Qwen3.6-35B-A3B": {"id": "Qwen/Qwen3.6-35B-A3B", "name": "Qwen3.6 35B-A3B", "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-17", "last_updated": "2026-04-17", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.15, "output": 0.95}}, "Qwen/Qwen3-Next-80B-A3B-Thinking": {"id": "Qwen/Qwen3-Next-80B-A3B-Thinking", "name": "Qwen3-Next-80B-A3B-Thinking", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-11", "last_updated": "2025-09-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 131072}, "cost": {"input": 0.3, "output": 2}}, "Qwen/Qwen3-Coder-480B-A35B-Instruct": {"id": "Qwen/Qwen3-Coder-480B-A35B-Instruct", "name": "Qwen3-Coder-480B-A35B-Instruct", "description": "Qwen coding model for software agents, repository edits, and code reasoning", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-23", "last_updated": "2025-07-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 66536}, "cost": {"input": 2, "output": 2}}, "Qwen/Qwen3-32B": {"id": "Qwen/Qwen3-32B", "name": "Qwen3 32B", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.29, "output": 0.59}}, "Qwen/Qwen3-Next-80B-A3B-Instruct": {"id": "Qwen/Qwen3-Next-80B-A3B-Instruct", "name": "Qwen3-Next-80B-A3B-Instruct", "description": "Qwen instruction model for multilingual chat, reasoning, and tool use", "family": "qwen", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-11", "last_updated": "2025-09-11", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 66536}, "cost": {"input": 0.25, "output": 1}}, "Qwen/Qwen3-235B-A22B-Thinking-2507": {"id": "Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "Qwen3-235B-A22B-Thinking-2507", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-25", "last_updated": "2025-07-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 131072}, "cost": {"input": 0.3, "output": 3}}, "MiniMaxAI/MiniMax-M2": {"id": "MiniMaxAI/MiniMax-M2", "name": "MiniMax-M2", "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "release_date": "2025-10-27", "last_updated": "2025-10-27", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 128000}, "cost": {"input": 0.3, "output": 1.2}}, "MiniMaxAI/MiniMax-M2.7": {"id": "MiniMaxAI/MiniMax-M2.7", "name": "MiniMax-M2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.06}}, "MiniMaxAI/MiniMax-M2.1": {"id": "MiniMaxAI/MiniMax-M2.1", "name": "MiniMax-M2.1", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2025-10", "release_date": "2025-12-23", "last_updated": "2025-12-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.3, "output": 1.2}}, "MiniMaxAI/MiniMax-M2.5": {"id": "MiniMaxAI/MiniMax-M2.5", "name": "MiniMax-M2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.3, "output": 1.2, "cache_read": 0.03}}, "MiniMaxAI/MiniMax-M3": {"id": "MiniMaxAI/MiniMax-M3", "name": "MiniMax-M3", "description": "MiniMax multimodal model for long-context coding, perception, and agent planning", "family": "minimax", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-06-01", "last_updated": "2026-06-01", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 524288, "output": 128000}, "cost": {"input": 0.3, "output": 1.2}}, "deepseek-ai/DeepSeek-V3": {"id": "deepseek-ai/DeepSeek-V3", "name": "DeepSeek-V3", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2024-12-26", "last_updated": "2024-12-26", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 64000, "output": 8192}, "cost": {"input": 0.4, "output": 1.3}}, "deepseek-ai/DeepSeek-V4-Flash-0731": {"id": "deepseek-ai/DeepSeek-V4-Flash-0731", "name": "DeepSeek V4 Flash 0731", "description": "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-07-31", "last_updated": "2026-07-31", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 384000}, "cost": {"input": 0.14, "output": 0.28}}, "deepseek-ai/DeepSeek-R1-0528": {"id": "deepseek-ai/DeepSeek-R1-0528", "name": "DeepSeek-R1-0528", "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-05-28", "last_updated": "2025-05-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 163840, "output": 163840}, "cost": {"input": 3, "output": 5}}, "deepseek-ai/DeepSeek-R1": {"id": "deepseek-ai/DeepSeek-R1", "name": "DeepSeek-R1", "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-01-20", "last_updated": "2025-05-29", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 64000, "output": 32768}, "cost": {"input": 0.7, "output": 2.5}}, "deepseek-ai/DeepSeek-V3.2": {"id": "deepseek-ai/DeepSeek-V3.2", "name": "DeepSeek-V3.2", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12-01", "last_updated": "2025-12-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 163840, "output": 65536}, "cost": {"input": 0.28, "output": 0.4}}, "deepseek-ai/DeepSeek-V3.1": {"id": "deepseek-ai/DeepSeek-V3.1", "name": "DeepSeek-V3.1", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-21", "last_updated": "2025-08-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 8192}, "cost": {"input": 0.27, "output": 1}}, "deepseek-ai/DeepSeek-V4-Flash": {"id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek V4 Flash", "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 384000}, "cost": {"input": 0.14, "output": 0.28}}, "deepseek-ai/DeepSeek-V4-Pro": {"id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", "description": "Open MoE flagship with million-token context for coding and long agent runs", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 393216}, "cost": {"input": 0.435, "output": 0.87, "cache_read": 0.003625}}, "stepfun-ai/Step-3.5-Flash": {"id": "stepfun-ai/Step-3.5-Flash", "name": "Step 3.5 Flash", "description": "StepFun flash lane for quick multimodal reasoning and coding assistance", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-29", "last_updated": "2026-02-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 256000}, "cost": {"input": 0.1, "output": 0.3}}, "stepfun-ai/Step-3.7-Flash": {"id": "stepfun-ai/Step-3.7-Flash", "name": "Step 3.7 Flash", "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "temperature": true, "knowledge": "2026-03-01", "release_date": "2026-05-29", "last_updated": "2026-05-29", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 256000}, "cost": {"input": 0.2, "output": 1.15}}, "moonshotai/Kimi-K2-Instruct": {"id": "moonshotai/Kimi-K2-Instruct", "name": "Kimi-K2-Instruct", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-07-14", "last_updated": "2025-07-14", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 1, "output": 3}}, "moonshotai/Kimi-K2.6": {"id": "moonshotai/Kimi-K2.6", "name": "Kimi-K2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.95, "output": 4, "cache_read": 0.16}}, "moonshotai/Kimi-K2.5": {"id": "moonshotai/Kimi-K2.5", "name": "Kimi-K2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2025-01", "release_date": "2026-01-01", "last_updated": "2026-01-01", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.6, "output": 3, "cache_read": 0.1}}, "moonshotai/Kimi-K2-Instruct-0905": {"id": "moonshotai/Kimi-K2-Instruct-0905", "name": "Kimi-K2-Instruct-0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-04", "last_updated": "2025-09-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 16384}, "cost": {"input": 1, "output": 3}}, "moonshotai/Kimi-K2.7-Code": {"id": "moonshotai/Kimi-K2.7-Code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.95, "output": 4}}, "moonshotai/Kimi-K2-Thinking": {"id": "moonshotai/Kimi-K2-Thinking", "name": "Kimi-K2-Thinking", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.6, "output": 2.5, "cache_read": 0.15}}, "moonshotai/Kimi-K3": {"id": "moonshotai/Kimi-K3", "name": "Kimi K3", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k3", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "high", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-07-16", "last_updated": "2026-07-16", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 131072}, "cost": {"input": 3, "output": 15}}, "openai/gpt-oss-20b": {"id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.1, "output": 0.5}}, "openai/gpt-oss-120b": {"id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2025-08-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}, "cost": {"input": 0.25, "output": 0.69}}, "meta-llama/Llama-3.3-70B-Instruct": {"id": "meta-llama/Llama-3.3-70B-Instruct", "name": "Llama-3.3-70B-Instruct", "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", "family": "llama", "attachment": false, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-12", "release_date": "2024-12-06", "last_updated": "2024-12-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 4096}, "cost": {"input": 0.59, "output": 0.79}}, "XiaomiMiMo/MiMo-V2.5-Pro": {"id": "XiaomiMiMo/MiMo-V2.5-Pro", "name": "MiMo-V2.5-Pro", "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 131072}, "cost": {"input": 1, "output": 3}}, "XiaomiMiMo/MiMo-V2.5": {"id": "XiaomiMiMo/MiMo-V2.5", "name": "MiMo-V2.5", "description": "MiMo model for long-context reasoning, perception, and agentic tasks", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-12", "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 131072}, "cost": {"input": 0.4, "output": 2}}, "XiaomiMiMo/MiMo-V2-Flash": {"id": "XiaomiMiMo/MiMo-V2-Flash", "name": "MiMo-V2-Flash", "description": "MiMo flash model for fast multimodal assistance and agent workflows", "family": "mimo", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "temperature": true, "knowledge": "2024-12", "release_date": "2025-12-16", "last_updated": "2025-12-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 4096}, "cost": {"input": 0.1, "output": 0.3}}}}, "anthropic": {"id": "anthropic", "env": ["ANTHROPIC_API_KEY"], "npm": "@ai-sdk/anthropic", "name": "Anthropic", "doc": "https://docs.anthropic.com/en/docs/about-claude/models", "models": {"claude-sonnet-4-6": {"id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "description": "Claude workhorse for coding agents, careful analysis, and production cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "max"]}, {"type": "budget_tokens", "min": 1024}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-02-17", "last_updated": "2026-03-13", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75}}, "claude-haiku-4-5": {"id": "claude-haiku-4-5", "name": "Claude Haiku 4.5 (latest)", "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 1024}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 64000}, "cost": {"input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25}}, "claude-opus-4-6": {"id": "claude-opus-4-6", "name": "Claude Opus 4.6", "description": "High-end Claude for difficult coding, planning, and slower expert reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "max"]}, {"type": "budget_tokens", "min": 1024}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05-31", "release_date": "2026-02-04", "last_updated": "2026-03-13", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "claude-fable-5": {"id": "claude-fable-5", "name": "Claude Fable 5", "description": "Claude model for creative writing, analysis, and controlled agent workflows", "family": "claude-fable", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-06-07", "last_updated": "2026-06-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5}}, "claude-opus-4-8": {"id": "claude-opus-4-8", "name": "Claude Opus 4.8", "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01", "release_date": "2026-05-28", "last_updated": "2026-05-28", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "experimental": {"modes": {"fast": {"cost": {"input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5}, "provider": {"body": {"speed": "fast"}, "headers": {"anthropic-beta": "fast-mode-2026-02-01"}}}}}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "claude-sonnet-4-5": {"id": "claude-sonnet-4-5", "name": "Claude Sonnet 4.5 (latest)", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 1024}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 64000}, "cost": {"input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75}}, "claude-sonnet-4-5-20250929": {"id": "claude-sonnet-4-5-20250929", "name": "Claude Sonnet 4.5", "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 1024}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-07-31", "release_date": "2025-09-29", "last_updated": "2025-09-29", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 64000}, "cost": {"input": 3, "output": 15, "cache_read": 0.3, "cache_write": 3.75}}, "claude-opus-4-5-20251101": {"id": "claude-opus-4-5-20251101", "name": "Claude Opus 4.5", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}, {"type": "budget_tokens", "min": 1024}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-01", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 64000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "claude-haiku-4-5-20251001": {"id": "claude-haiku-4-5-20251001", "name": "Claude Haiku 4.5", "description": "Fast Claude model for responsive assistance, classification, and lightweight agents", "family": "claude-haiku", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "budget_tokens", "min": 1024}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-02-28", "release_date": "2025-10-15", "last_updated": "2025-10-15", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 64000}, "cost": {"input": 1, "output": 5, "cache_read": 0.1, "cache_write": 1.25}}, "claude-opus-4-7": {"id": "claude-opus-4-7", "name": "Claude Opus 4.7", "description": "Stronger Opus tier for advanced software work and high-stakes reasoning", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-04-14", "last_updated": "2026-04-16", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "claude-opus-4-5": {"id": "claude-opus-4-5", "name": "Claude Opus 4.5 (latest)", "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}, {"type": "budget_tokens", "min": 1024}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2025-11-24", "last_updated": "2025-11-24", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 64000}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}, "claude-sonnet-5": {"id": "claude-sonnet-5", "name": "Claude Sonnet 5", "description": "Everyday Claude agent model for coding, planning, browsing, and general work", "family": "claude-sonnet", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-01-31", "release_date": "2026-06-29", "last_updated": "2026-06-30", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "cost": {"input": 2, "output": 10, "cache_read": 0.2, "cache_write": 2.5}}, "claude-opus-5": {"id": "claude-opus-5", "name": "Claude Opus 5", "description": "Strongest Claude Opus model for coding, agents, and professional work", "family": "claude-opus", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-05", "release_date": "2026-07-24", "last_updated": "2026-07-24", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1000000, "output": 128000}, "experimental": {"modes": {"fast": {"cost": {"input": 10, "output": 50, "cache_read": 1, "cache_write": 12.5}, "provider": {"body": {"speed": "fast"}, "headers": {"anthropic-beta": "fast-mode-2026-02-01"}}}}}, "cost": {"input": 5, "output": 25, "cache_read": 0.5, "cache_write": 6.25}}}}, "chutes": {"id": "chutes", "env": ["CHUTES_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://llm.chutes.ai/v1", "name": "Chutes", "doc": "https://llm.chutes.ai/v1/models", "models": {"Nemotron-3-Nano-Omni-30B-TEE": {"id": "Nemotron-3-Nano-Omni-30B-TEE", "name": "Nemotron 3 Nano Omni 30B TEE", "description": "Omni-modal model for text, vision, audio, and multimodal agent tasks", "family": "nemotron", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "release_date": "2026-07-23", "last_updated": "2026-07-23", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 0}, "cost": {"input": 0.0245, "output": 0.0978, "cache_read": 0.0024499999999999995}}, "google/gemma-4-31B-turbo-TEE": {"id": "google/gemma-4-31B-turbo-TEE", "name": "gemma 4 31B turbo TEE", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-04-02", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 0.12, "output": 0.37, "cache_read": 0.011999999999999997}}, "zai-org/GLM-5.1-TEE": {"id": "zai-org/GLM-5.1-TEE", "name": "GLM 5.1 TEE", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 202752, "output": 65535}, "cost": {"input": 0.98, "output": 3.08, "cache_read": 0.09799999999999998}}, "zai-org/GLM-5.2-TEE": {"id": "zai-org/GLM-5.2-TEE", "name": "GLM 5.2 TEE", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 65535}, "cost": {"input": 1.25, "output": 3.95, "cache_read": 0.12499999999999997}}, "unsloth/Mistral-Nemo-Instruct-2407-TEE": {"id": "unsloth/Mistral-Nemo-Instruct-2407-TEE", "name": "Mistral Nemo Instruct 2407 TEE", "description": "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment", "family": "mistral-nemo", "attachment": false, "reasoning": false, "tool_call": false, "temperature": true, "knowledge": "2024-07", "release_date": "2024-07-01", "last_updated": "2024-07-01", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 131072}, "cost": {"input": 0.0245, "output": 0.0978, "cache_read": 0.0024499999999999995}}, "Qwen/Qwen3-235B-A22B-Thinking-2507-TEE": {"id": "Qwen/Qwen3-235B-A22B-Thinking-2507-TEE", "name": "Qwen3 235B A22B Thinking 2507 TEE", "description": "Qwen reasoning model for deliberate problem solving, math, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07", "last_updated": "2026-06-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.2989, "output": 1.1957, "cache_read": 0.029889999999999993}}, "Qwen/Qwen3-32B-TEE": {"id": "Qwen/Qwen3-32B-TEE", "name": "Qwen3 32B TEE", "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding", "family": "qwen", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-04", "last_updated": "2025-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 40960, "output": 40960}, "cost": {"input": 0.104, "output": 0.416, "cache_read": 0.010399999999999998}}, "Qwen/Qwen3.6-27B-TEE": {"id": "Qwen/Qwen3.6-27B-TEE", "name": "Qwen3.6 27B TEE", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-22", "last_updated": "2026-04-22", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.3, "output": 2, "cache_read": 0.029999999999999992}}, "Qwen/Qwen3.5-397B-A17B-TEE": {"id": "Qwen/Qwen3.5-397B-A17B-TEE", "name": "Qwen3.5 397B A17B TEE", "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-02-15", "last_updated": "2026-02-15", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}, "cost": {"input": 0.45, "output": 3, "cache_read": 0.04499999999999999}}, "deepseek-ai/DeepSeek-V3.2-TEE": {"id": "deepseek-ai/DeepSeek-V3.2-TEE", "name": "DeepSeek V3.2 TEE", "description": "DeepSeek chat model for instruction following, coding, and analysis", "family": "deepseek", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2024-07", "release_date": "2025-12", "last_updated": "2026-06-21", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 65536}, "cost": {"input": 1, "output": 1, "cache_read": 0.09999999999999998}}, "deepseek-ai/DeepSeek-V4-Flash-0731-TEE": {"id": "deepseek-ai/DeepSeek-V4-Flash-0731-TEE", "name": "DeepSeek V4 Flash 0731 TEE", "description": "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-07-31", "last_updated": "2026-08-02", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 131072}, "cost": {"input": 0.14, "output": 0.28, "cache_read": 0.013999999999999999}}, "moonshotai/Kimi-K2.6-TEE": {"id": "moonshotai/Kimi-K2.6-TEE", "name": "Kimi K2.6 TEE", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-12", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65535}, "cost": {"input": 0.58, "output": 3.4, "cache_read": 0.05799999999999998}}, "moonshotai/Kimi-K3-TEE": {"id": "moonshotai/Kimi-K3-TEE", "name": "Kimi K3 TEE", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k3", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-07-29", "last_updated": "2026-07-29", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 65535}, "cost": {"input": 3, "output": 15, "cache_read": 0.29999999999999993}}, "z-image-turbo": {"name": "Z Image Turbo", "modalities": {"input": ["text"], "output": ["image"]}, "id": "z-image-turbo"}, "Qwen-Image-2512": {"modalities": {"input": ["text"], "output": ["image"]}, "id": "Qwen-Image-2512", "name": "Qwen Image 2512"}, "Qwen-Image-Edit-2511": {"name": "Qwen Image Edit", "modalities": {"input": ["text", "image"], "output": ["image"]}, "id": "Qwen-Image-Edit-2511"}, "flux": {"name": "FLUX", "modalities": {"input": ["text"], "output": ["image"]}, "id": "flux"}, "dreamshaper": {"name": "Dreamshaper", "modalities": {"input": ["text"], "output": ["image"]}, "id": "dreamshaper"}, "juggernaut": {"name": "Juggernaut", "modalities": {"input": ["text"], "output": ["image"]}, "id": "juggernaut"}, "ilustmix": {"name": "Ilustmix", "modalities": {"input": ["text"], "output": ["image"]}, "id": "ilustmix"}}}, "cerebras": {"id": "cerebras", "env": ["CEREBRAS_API_KEY"], "npm": "@ai-sdk/cerebras", "name": "Cerebras", "doc": "https://inference-docs.cerebras.ai/models/overview", "models": {"gemma-4-31b": {"id": "gemma-4-31b", "name": "Gemma 4 31B IT", "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-04-02", "last_updated": "2026-07-01", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 40960}, "status": "beta", "cost": {"input": 0.99, "output": 1.49}}, "zai-glm-4.7": {"id": "zai-glm-4.7", "name": "Z.AI GLM-4.7", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2026-01-07", "last_updated": "2026-06-10", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 40960}, "status": "beta", "cost": {"input": 2.25, "output": 2.75, "cache_read": 2.25, "cache_write": 0}}, "gpt-oss-120b": {"id": "gpt-oss-120b", "name": "GPT OSS 120B", "description": "Open GPT reasoning model for self-hosted agents and controllable deployments", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": true, "release_date": "2025-08-05", "last_updated": "2026-06-10", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 40960}, "cost": {"input": 0.35, "output": 0.75}}}}, "ollama-cloud": {"id": "ollama-cloud", "env": ["OLLAMA_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://ollama.com/v1", "name": "Ollama Cloud", "doc": "https://docs.ollama.com/cloud", "models": {"glm-5.1": {"id": "glm-5.1", "name": "glm-5.1", "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "release_date": "2026-03-27", "last_updated": "2026-04-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 202752, "output": 131072}}, "nemotron-3-super": {"id": "nemotron-3-super", "name": "nemotron-3-super", "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "release_date": "2026-03-11", "last_updated": "2026-03-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}}, "deepseek-v4-flash": {"id": "deepseek-v4-flash", "name": "deepseek-v4-flash", "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "max"]}], "tool_call": true, "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 1048576}}, "kimi-k2.5": {"id": "kimi-k2.5", "name": "kimi-k2.5", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "release_date": "2026-01-27", "last_updated": "2026-01-27", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}}, "minimax-m2.7": {"id": "minimax-m2.7", "name": "minimax-m2.7", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "release_date": "2026-03-18", "last_updated": "2026-03-18", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 196608, "output": 196608}}, "nemotron-3-nano:30b": {"id": "nemotron-3-nano:30b", "name": "nemotron-3-nano:30b", "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "release_date": "2025-12-15", "last_updated": "2026-01-19", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 131072}}, "mistral-large-3:675b": {"id": "mistral-large-3:675b", "name": "mistral-large-3:675b", "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work", "family": "mistral-large", "attachment": true, "reasoning": false, "tool_call": true, "release_date": "2025-12-02", "last_updated": "2026-01-19", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}}, "glm-5.2": {"id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 976000, "output": 131072}}, "nemotron-3-ultra": {"id": "nemotron-3-ultra", "name": "nemotron-3-ultra", "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", "family": "nemotron", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "release_date": "2026-06-04", "last_updated": "2026-06-04", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 128000}}, "kimi-k2.6": {"id": "kimi-k2.6", "name": "kimi-k2.6", "description": "Kimi multimodal agent model for visual understanding, coding, and planning", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "release_date": "2026-04-20", "last_updated": "2026-04-20", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}}, "qwen3.5:397b": {"id": "qwen3.5:397b", "name": "qwen3.5:397b", "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks", "family": "qwen", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_details"}, "release_date": "2026-02-15", "last_updated": "2026-02-17", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 65536}}, "minimax-m3": {"id": "minimax-m3", "name": "minimax-m3", "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks", "family": "minimax-m3", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "medium", "high", "max"]}], "tool_call": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-05-31", "last_updated": "2026-05-31", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 512000, "output": 131072}}, "deepseek-v4-pro": {"id": "deepseek-v4-pro", "name": "deepseek-v4-pro", "description": "Flagship DeepSeek model for coding, reasoning, and agentic work", "family": "deepseek-thinking", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "max"]}], "tool_call": true, "release_date": "2026-04-24", "last_updated": "2026-04-24", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 1048576}}, "gpt-oss:120b": {"id": "gpt-oss:120b", "name": "gpt-oss:120b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "release_date": "2025-08-05", "last_updated": "2026-01-19", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}}, "deepseek-v4-flash:0731": {"id": "deepseek-v4-flash:0731", "name": "DeepSeek V4 Flash 0731", "description": "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding", "family": "deepseek-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["high", "max"]}], "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-05", "release_date": "2026-07-31", "last_updated": "2026-07-31", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 1048576}}, "gemma4:31b": {"id": "gemma4:31b", "name": "gemma4:31b", "description": "Open Gemma instruction model for efficient chat and self-hosted deployments", "family": "gemma", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "knowledge": "2025-01", "release_date": "2026-04-02", "last_updated": "2026-04-08", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}}, "minimax-m2.5": {"id": "minimax-m2.5", "name": "minimax-m2.5", "description": "MiniMax model for chat, coding, office work, and agentic tasks", "family": "minimax", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "knowledge": "2025-01", "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}}, "kimi-k2.7-code": {"id": "kimi-k2.7-code", "name": "kimi-k2.7-code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}}, "kimi-k3": {"id": "kimi-k3", "name": "kimi-k3", "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work", "family": "kimi-k3", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "high", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "release_date": "2026-07-16", "last_updated": "2026-07-27", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 131072}}, "gpt-oss:20b": {"id": "gpt-oss:20b", "name": "gpt-oss:20b", "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads", "family": "gpt-oss", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "release_date": "2025-08-05", "last_updated": "2026-01-19", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 32768}}}}, "moonshotai": {"id": "moonshotai", "env": ["MOONSHOT_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.moonshot.ai/v1", "name": "Moonshot AI", "doc": "https://platform.moonshot.ai/docs/api/chat", "models": {"kimi-k2.5": {"id": "kimi-k2.5", "name": "Kimi K2.5", "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", "family": "kimi-k2", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-01", "last_updated": "2026-01", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.6, "output": 3, "cache_read": 0.1}}, "kimi-k2-thinking-turbo": {"id": "kimi-k2-thinking-turbo", "name": "Kimi K2 Thinking Turbo", "description": "Kimi reasoning model for long-horizon research, planning, and tool use", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 1.15, "output": 8, "cache_read": 0.15}}, "kimi-k2-0711-preview": {"id": "kimi-k2-0711-preview", "name": "Kimi K2 0711", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-07-14", "last_updated": "2025-07-14", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 16384}, "cost": {"input": 0.6, "output": 2.5, "cache_read": 0.15}}, "kimi-k2.7-code-highspeed": {"id": "kimi-k2.7-code-highspeed", "name": "Kimi K2.7 Code HighSpeed", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 1.9, "output": 8, "cache_read": 0.38}}, "kimi-k2.6": {"id": "kimi-k2.6", "name": "Kimi K2.6", "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "knowledge": "2025-01", "release_date": "2026-04-21", "last_updated": "2026-04-21", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.95, "output": 4, "cache_read": 0.16}}, "kimi-k2-turbo-preview": {"id": "kimi-k2-turbo-preview", "name": "Kimi K2 Turbo", "description": "Fast Kimi model for responsive chat, coding help, and agent loops", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 2.4, "output": 10, "cache_read": 0.6}}, "kimi-k2-0905-preview": {"id": "kimi-k2-0905-preview", "name": "Kimi K2 0905", "description": "Kimi model for long-context chat, coding, and agentic reasoning", "family": "kimi-k2", "attachment": false, "reasoning": false, "tool_call": true, "temperature": true, "knowledge": "2024-10", "release_date": "2025-09-05", "last_updated": "2025-09-05", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.6, "output": 2.5, "cache_read": 0.15}}, "kimi-k2.7-code": {"id": "kimi-k2.7-code", "name": "Kimi K2.7 Code", "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", "family": "kimi-k2", "attachment": true, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": false, "knowledge": "2025-01", "release_date": "2026-06-12", "last_updated": "2026-06-12", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.95, "output": 4, "cache_read": 0.19}}, "kimi-k2-thinking": {"id": "kimi-k2-thinking", "name": "Kimi K2 Thinking", "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions", "family": "kimi-thinking", "attachment": false, "reasoning": true, "reasoning_options": [], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2024-08", "release_date": "2025-11-06", "last_updated": "2025-11-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 262144, "output": 262144}, "cost": {"input": 0.6, "output": 2.5, "cache_read": 0.15}}, "kimi-k3": {"id": "kimi-k3", "name": "Kimi K3", "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work", "family": "kimi-k3", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}, {"type": "effort", "values": ["low", "high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": false, "release_date": "2026-07-16", "last_updated": "2026-07-16", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1048576, "output": 131072}, "cost": {"input": 3, "output": 15, "cache_read": 0.3}}}}, "openai": {"id": "openai", "env": ["OPENAI_API_KEY"], "npm": "@ai-sdk/openai", "name": "OpenAI", "doc": "https://platform.openai.com/docs/models", "models": {"gpt-image-2": {"name": "GPT Image 2", "release_date": "2026-04-22", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 8.0, "output": 15.0}, "id": "gpt-image-2"}, "gpt-5.2-pro": {"id": "gpt-5.2-pro", "name": "GPT-5.2 Pro", "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["medium", "high", "xhigh"]}], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 21, "output": 168}}, "gpt-5.5-pro": {"id": "gpt-5.5-pro", "name": "GPT-5.5 Pro", "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 30, "output": 180, "tiers": [{"input": 60, "output": 270, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 60, "output": 270}}}, "gpt-4.1-mini": {"id": "gpt-4.1-mini", "name": "GPT-4.1 mini", "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1047576, "output": 32768}, "cost": {"input": 0.4, "output": 1.6, "cache_read": 0.1}}, "gpt-4o": {"id": "gpt-4o", "name": "GPT-4o", "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-08-06", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 2.5, "output": 10, "cache_read": 1.25}}, "gpt-5.3-codex-spark": {"id": "gpt-5.3-codex-spark", "name": "GPT-5.3 Codex Spark", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex-spark", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "input": 100000, "output": 32000}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "gpt-5.4-pro": {"id": "gpt-5.4-pro", "name": "GPT-5.4 Pro", "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["medium", "high", "xhigh"]}], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "cost": {"input": 30, "output": 180, "tiers": [{"input": 60, "output": 270, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 60, "output": 270}}}, "gpt-5.6-sol": {"id": "gpt-5.6-sol", "name": "GPT-5.6 Sol", "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", "family": "gpt-sol", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "experimental": {"modes": {"fast": {"cost": {"input": 10, "output": 60, "cache_read": 1, "cache_write": 12.5}, "provider": {"body": {"service_tier": "priority"}}}, "pro": {"provider": {"body": {"reasoning": {"mode": "pro"}}}}}}, "cost": {"input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25, "tiers": [{"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5}}}, "text-embedding-ada-002": {"id": "text-embedding-ada-002", "name": "text-embedding-ada-002", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2022-12", "release_date": "2022-12-15", "last_updated": "2022-12-15", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 8192, "output": 1536}, "cost": {"input": 0.1, "output": 0}}, "gpt-4.1-nano": {"id": "gpt-4.1-nano", "name": "GPT-4.1 nano", "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", "family": "gpt-nano", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1047576, "output": 32768}, "status": "deprecated", "cost": {"input": 0.1, "output": 0.4, "cache_read": 0.025}}, "gpt-5.2-chat-latest": {"id": "gpt-5.2-chat-latest", "name": "GPT-5.2 Chat", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["medium"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "o3-mini": {"id": "o3-mini", "name": "o3-mini", "description": "Smaller o-series reasoner for economical coding, math, and planning tasks", "family": "o-mini", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2024-12-20", "last_updated": "2025-01-29", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "status": "deprecated", "cost": {"input": 1.1, "output": 4.4, "cache_read": 0.55}}, "gpt-image-1-mini": {"name": "GPT Image 1 Mini", "release_date": "2025-10-16", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 5.0, "output": 8.0}, "id": "gpt-image-1-mini"}, "gpt-5.5": {"id": "gpt-5.5", "name": "GPT-5.5", "description": "Default frontier GPT for coding, computer use, research, and knowledge work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-12-01", "release_date": "2026-04-23", "last_updated": "2026-04-23", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "experimental": {"modes": {"fast": {"cost": {"input": 12.5, "output": 75, "cache_read": 1.25}, "provider": {"body": {"service_tier": "priority"}}}}}, "cost": {"input": 5, "output": 30, "cache_read": 0.5, "tiers": [{"input": 10, "output": 45, "cache_read": 1, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 10, "output": 45, "cache_read": 1}}}, "o3-pro": {"id": "o3-pro", "name": "o3-pro", "description": "High-effort o3 tier for difficult technical reasoning and careful answers", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-06-10", "last_updated": "2025-06-10", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "cost": {"input": 20, "output": 80}}, "gpt-4o-mini": {"id": "gpt-4o-mini", "name": "GPT-4o mini", "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic", "family": "gpt-mini", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-07-18", "last_updated": "2024-07-18", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 0.15, "output": 0.6, "cache_read": 0.075}}, "chatgpt-image-latest": {"name": "ChatGPT Image Latest", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 5.0, "output": 32.0}, "id": "chatgpt-image-latest"}, "gpt-5": {"id": "gpt-5", "name": "GPT-5", "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.25, "output": 10, "cache_read": 0.125}}, "gpt-realtime-2.1": {"id": "gpt-realtime-2.1", "name": "GPT-Realtime-2.1", "description": "Realtime speech-to-speech model with configurable reasoning, tool use, and robust voice-agent behavior", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": false, "temperature": false, "knowledge": "2024-09-30", "release_date": "2026-07-06", "last_updated": "2026-07-06", "modalities": {"input": ["text", "audio", "image"], "output": ["text", "audio"]}, "open_weights": false, "limit": {"context": 128000, "input": 96000, "output": 32000}, "cost": {"input": 4, "output": 24, "cache_read": 0.4, "input_audio": 32, "output_audio": 64}}, "gpt-3.5-turbo": {"id": "gpt-3.5-turbo", "name": "GPT-3.5-turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": false, "reasoning": false, "tool_call": false, "structured_output": false, "temperature": true, "knowledge": "2021-09-01", "release_date": "2023-03-01", "last_updated": "2023-11-06", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 16385, "output": 4096}, "status": "deprecated", "cost": {"input": 0.5, "output": 1.5, "cache_read": 0}}, "gpt-4o-2024-05-13": {"id": "gpt-4o-2024-05-13", "name": "GPT-4o (2024-05-13)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-05-13", "last_updated": "2024-05-13", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 4096}, "status": "deprecated", "cost": {"input": 5, "output": 15}}, "gpt-4o-2024-11-20": {"id": "gpt-4o-2024-11-20", "name": "GPT-4o (2024-11-20)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-11-20", "last_updated": "2024-11-20", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 2.5, "output": 10, "cache_read": 1.25}}, "gpt-5.4": {"id": "gpt-5.4", "name": "GPT-5.4", "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-05", "last_updated": "2026-03-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "experimental": {"modes": {"fast": {"cost": {"input": 5, "output": 30, "cache_read": 0.5}, "provider": {"body": {"service_tier": "priority"}}}}}, "cost": {"input": 2.5, "output": 15, "cache_read": 0.25, "tiers": [{"input": 5, "output": 22.5, "cache_read": 0.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 5, "output": 22.5, "cache_read": 0.5}}}, "gpt-5.3-chat-latest": {"id": "gpt-5.3-chat-latest", "name": "GPT-5.3 Chat (latest)", "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2025-08-31", "release_date": "2026-03-03", "last_updated": "2026-03-03", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "gpt-image-1.5": {"name": "GPT Image 1.5", "release_date": "2025-10-16", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 5.0, "output": 32.0}, "id": "gpt-image-1.5"}, "gpt-5.4-nano": {"id": "gpt-5.4-nano", "name": "GPT-5.4 nano", "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 0.2, "output": 1.25, "cache_read": 0.02}}, "gpt-5-pro": {"id": "gpt-5-pro", "name": "GPT-5 Pro", "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", "family": "gpt-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-10-06", "last_updated": "2025-10-06", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 272000}, "cost": {"input": 15, "output": 120}}, "gpt-5.4-mini": {"id": "gpt-5.4-mini", "name": "GPT-5.4 mini", "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-03-17", "last_updated": "2026-03-17", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "experimental": {"modes": {"fast": {"cost": {"input": 1.5, "output": 9, "cache_read": 0.15}, "provider": {"body": {"service_tier": "priority"}}}}}, "cost": {"input": 0.75, "output": 4.5, "cache_read": 0.075}}, "o1": {"id": "o1", "name": "o1", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2024-12-05", "last_updated": "2024-12-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "status": "deprecated", "cost": {"input": 15, "output": 60, "cache_read": 7.5}}, "gpt-5.6-luna": {"id": "gpt-5.6-luna", "name": "GPT-5.6 Luna", "description": "Cost-efficient GPT-5.6 model for fast, high-volume workloads", "family": "gpt-luna", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "experimental": {"modes": {"fast": {"cost": {"input": 0.4, "output": 2.4, "cache_read": 0.04, "cache_write": 0.5}, "provider": {"body": {"service_tier": "priority"}}}, "pro": {"provider": {"body": {"reasoning": {"mode": "pro"}}}}}}, "cost": {"input": 0.2, "output": 1.2, "cache_read": 0.02, "cache_write": 0.25, "tiers": [{"input": 0.4, "output": 1.8, "cache_read": 0.04, "cache_write": 0.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 0.4, "output": 1.8, "cache_read": 0.04, "cache_write": 0.5}}}, "gpt-5.2": {"id": "gpt-5.2", "name": "GPT-5.2", "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2025-12-11", "last_updated": "2025-12-11", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "gpt-5.3-codex": {"id": "gpt-5.3-codex", "name": "GPT-5.3 Codex", "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work", "family": "gpt-codex", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2025-08-31", "release_date": "2026-02-05", "last_updated": "2026-02-05", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.75, "output": 14, "cache_read": 0.175}}, "gpt-5-mini": {"id": "gpt-5-mini", "name": "GPT-5 Mini", "description": "Small GPT-5 for responsive agents, coding help, and everyday automation", "family": "gpt-mini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 0.25, "output": 2, "cache_read": 0.025}}, "o1-pro": {"id": "o1-pro", "name": "o1-pro", "description": "O-series reasoning model for hard analysis, math, coding, and planning", "family": "o-pro", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2023-09", "release_date": "2025-03-19", "last_updated": "2025-03-19", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "status": "deprecated", "cost": {"input": 150, "output": 600}}, "gpt-5.6": {"id": "gpt-5.6", "name": "GPT-5.6", "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", "family": "gpt-sol", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "experimental": {"modes": {"fast": {"cost": {"input": 10, "output": 60, "cache_read": 1, "cache_write": 12.5}, "provider": {"body": {"service_tier": "priority"}}}, "pro": {"provider": {"body": {"reasoning": {"mode": "pro"}}}}}}, "cost": {"input": 5, "output": 30, "cache_read": 0.5, "cache_write": 6.25, "tiers": [{"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 10, "output": 45, "cache_read": 1, "cache_write": 12.5}}}, "gpt-5.1": {"id": "gpt-5.1", "name": "GPT-5.1", "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-09-30", "release_date": "2025-11-13", "last_updated": "2025-11-13", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 1.25, "output": 10, "cache_read": 0.125}}, "gpt-4-turbo": {"id": "gpt-4-turbo", "name": "GPT-4 Turbo", "description": "Compact GPT model for low-latency assistance and high-volume workloads", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-12", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 4096}, "status": "deprecated", "cost": {"input": 10, "output": 30}}, "gpt-4o-2024-08-06": {"id": "gpt-4o-2024-08-06", "name": "GPT-4o (2024-08-06)", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2023-09", "release_date": "2024-08-06", "last_updated": "2024-08-06", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 128000, "output": 16384}, "cost": {"input": 2.5, "output": 10, "cache_read": 1.25}}, "gpt-5-nano": {"id": "gpt-5-nano", "name": "GPT-5 Nano", "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", "family": "gpt-nano", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["minimal", "low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05-30", "release_date": "2025-08-07", "last_updated": "2025-08-07", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 400000, "input": 272000, "output": 128000}, "cost": {"input": 0.05, "output": 0.4, "cache_read": 0.005}}, "o3": {"id": "o3", "name": "o3", "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", "family": "o", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "cost": {"input": 2, "output": 8, "cache_read": 0.5}}, "gpt-5.6-terra": {"id": "gpt-5.6-terra", "name": "GPT-5.6 Terra", "description": "Balanced GPT-5.6 model for capable, cost-efficient everyday work", "family": "gpt-terra", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["none", "low", "medium", "high", "xhigh", "max"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2026-02-16", "release_date": "2026-07-09", "last_updated": "2026-07-09", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1050000, "input": 922000, "output": 128000}, "experimental": {"modes": {"fast": {"cost": {"input": 4, "output": 24, "cache_read": 0.4, "cache_write": 5}, "provider": {"body": {"service_tier": "priority"}}}, "pro": {"provider": {"body": {"reasoning": {"mode": "pro"}}}}}}, "cost": {"input": 2, "output": 12, "cache_read": 0.2, "cache_write": 2.5, "tiers": [{"input": 4, "output": 18, "cache_read": 0.4, "cache_write": 5, "tier": {"type": "context", "size": 272000}}], "context_over_200k": {"input": 4, "output": 18, "cache_read": 0.4, "cache_write": 5}}}, "gpt-image-1": {"name": "GPT Image 1", "release_date": "2025-10-14", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 5.0, "output": 40.0}, "id": "gpt-image-1"}, "gpt-4.1": {"id": "gpt-4.1", "name": "GPT-4.1", "description": "Long-lived GPT workhorse for coding, instruction following, and production apps", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": true, "temperature": true, "knowledge": "2024-04", "release_date": "2025-04-14", "last_updated": "2025-04-14", "modalities": {"input": ["text", "image", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 1047576, "output": 32768}, "cost": {"input": 2, "output": 8, "cache_read": 0.5}}, "o4-mini": {"id": "o4-mini", "name": "o4-mini", "description": "Fast o-series model for compact reasoning, coding, and tool use", "family": "o-mini", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["low", "medium", "high"]}], "tool_call": true, "structured_output": true, "temperature": false, "knowledge": "2024-05", "release_date": "2025-04-16", "last_updated": "2025-04-16", "modalities": {"input": ["text", "image"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 100000}, "status": "deprecated", "cost": {"input": 1.1, "output": 4.4, "cache_read": 0.275}}, "gpt-4": {"id": "gpt-4", "name": "GPT-4", "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks", "family": "gpt", "attachment": true, "reasoning": false, "tool_call": true, "structured_output": false, "temperature": true, "knowledge": "2023-11", "release_date": "2023-11-06", "last_updated": "2024-04-09", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 8192, "output": 8192}, "status": "deprecated", "cost": {"input": 30, "output": 60}}, "text-embedding-3-large": {"id": "text-embedding-3-large", "name": "text-embedding-3-large", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-01", "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 8191, "output": 3072}, "cost": {"input": 0.13, "output": 0}}, "text-embedding-3-small": {"id": "text-embedding-3-small", "name": "text-embedding-3-small", "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", "family": "text-embedding", "attachment": false, "reasoning": false, "tool_call": false, "temperature": false, "knowledge": "2024-01", "release_date": "2024-01-25", "last_updated": "2024-01-25", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 8191, "output": 1536}, "cost": {"input": 0.02, "output": 0}}}}, "zai": {"id": "zai", "env": ["ZHIPU_API_KEY"], "npm": "@ai-sdk/openai-compatible", "api": "https://api.z.ai/api/paas/v4", "name": "Z.AI", "doc": "https://docs.z.ai/guides/overview/pricing", "models": {"glm-4.6v": {"id": "glm-4.6v", "name": "GLM-4.6V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-08", "last_updated": "2025-12-08", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 128000, "output": 32768}, "cost": {"input": 0.3, "output": 0.9}}, "glm-5": {"id": "glm-5", "name": "GLM-5", "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-02-12", "last_updated": "2026-02-12", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 1, "output": 3.2, "cache_read": 0.2, "cache_write": 0}}, "glm-4.5-air": {"id": "glm-4.5-air", "name": "GLM-4.5-Air", "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", "family": "glm-air", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 98304}, "cost": {"input": 0.2, "output": 1.1, "cache_read": 0.03, "cache_write": 0}}, "glm-5.1": {"id": "glm-5.1", "name": "GLM-5.1", "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-04-07", "last_updated": "2026-04-07", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 200000, "output": 131072}, "cost": {"input": 1.4, "output": 4.4, "cache_read": 0.26, "cache_write": 0}}, "glm-4.7-flash": {"id": "glm-4.7-flash", "name": "GLM-4.7-Flash", "description": "Budget GLM lane for fast coding help, routing, and everyday automation", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 200000, "output": 131072}, "cost": {"input": 0, "output": 0, "cache_read": 0, "cache_write": 0}}, "glm-5.2": {"id": "glm-5.2", "name": "GLM-5.2", "description": "Open flagship GLM for long-horizon coding agents and million-token context work", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "effort", "values": ["high", "max"]}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-06-13", "last_updated": "2026-06-13", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 1000000, "output": 131072}, "cost": {"input": 1.4, "output": 4.4, "cache_read": 0.26, "cache_write": 0}}, "glm-4.7-flashx": {"id": "glm-4.7-flashx", "name": "GLM-4.7-FlashX", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2026-01-19", "last_updated": "2026-01-19", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 200000, "output": 131072}, "cost": {"input": 0.07, "output": 0.4, "cache_read": 0.01, "cache_write": 0}}, "glm-4.6": {"id": "glm-4.6", "name": "GLM-4.6", "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-09-30", "last_updated": "2025-09-30", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0}}, "glm-4.5": {"id": "glm-4.5", "name": "GLM-4.5", "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 98304}, "cost": {"input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0}}, "glm-4.5v": {"id": "glm-4.5v", "name": "GLM-4.5V", "description": "GLM vision model for visual reasoning, documents, and multimodal agents", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-08-11", "last_updated": "2025-08-11", "modalities": {"input": ["text", "image", "video"], "output": ["text"]}, "open_weights": true, "limit": {"context": 64000, "output": 16384}, "cost": {"input": 0.6, "output": 1.8}}, "glm-4.7": {"id": "glm-4.7", "name": "GLM-4.7", "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "knowledge": "2025-04", "release_date": "2025-12-22", "last_updated": "2025-12-22", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 204800, "output": 131072}, "cost": {"input": 0.6, "output": 2.2, "cache_read": 0.11, "cache_write": 0}}, "glm-5-turbo": {"id": "glm-5-turbo", "name": "GLM-5-Turbo", "description": "Faster GLM-5 lane for coding agents that need lower latency", "family": "glm", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "structured_output": true, "temperature": true, "release_date": "2026-03-16", "last_updated": "2026-03-16", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 131072}, "cost": {"input": 1.2, "output": 4, "cache_read": 0.24, "cache_write": 0}}, "glm-5v-turbo": {"id": "glm-5v-turbo", "name": "GLM-5V-Turbo", "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", "family": "glm", "attachment": true, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "interleaved": {"field": "reasoning_content"}, "temperature": true, "release_date": "2026-04-01", "last_updated": "2026-04-01", "modalities": {"input": ["text", "image", "video", "pdf"], "output": ["text"]}, "open_weights": false, "limit": {"context": 200000, "output": 131072}, "cost": {"input": 1.2, "output": 4, "cache_read": 0.24, "cache_write": 0}}, "glm-4.5-flash": {"id": "glm-4.5-flash", "name": "GLM-4.5-Flash", "description": "Efficient GLM model for fast reasoning, coding, and agent workflows", "family": "glm-flash", "attachment": false, "reasoning": true, "reasoning_options": [{"type": "toggle"}], "tool_call": true, "temperature": true, "knowledge": "2025-04", "release_date": "2025-07-28", "last_updated": "2025-07-28", "modalities": {"input": ["text"], "output": ["text"]}, "open_weights": true, "limit": {"context": 131072, "output": 98304}, "cost": {"input": 0, "output": 0, "cache_read": 0, "cache_write": 0}}, "glm-image": {"name": "GLM-Image", "modalities": {"input": ["text"], "output": ["image"]}, "cost": {"input": 0, "output": 0.015}, "id": "glm-image"}}}}