{ "generatedAt": "2026-08-19T00:56:30.963Z", "providers": [ { "provider": "openai", "source": { "url": "https://developers.openai.com/api/docs/pricing.md", "fetchedAt": "2026-08-19T00:56:00.889Z" }, "models": [ { "id": "gpt-5.6-sol", "name": "gpt-5.6-sol", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$45.00" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$45.00" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.25" }, { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50" }, { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.25" }, { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$22.50" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [ { "amount": 3.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.125" }, { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.25" } ], "other": [] }, { "name": "flex", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$22.50" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [ { "amount": 3.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.125" }, { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.25" } ], "other": [] }, { "name": "fast", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$20.00" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$60.00" }, { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$90.00" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50" }, { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25.00" } ], "other": [] } ] }, { "id": "gpt-5.6-terra", "name": "gpt-5.6-terra", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.00" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.00" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$18.00" } ], "cacheRead": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" }, { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.00" }, { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$9.00" } ], "cacheRead": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10" }, { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" } ], "other": [] }, { "name": "flex", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.00" }, { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$9.00" } ], "cacheRead": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10" }, { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" } ], "other": [] }, { "name": "fast", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.00" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$8.00" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$24.00" }, { "amount": 36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$36.00" } ], "cacheRead": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" }, { "amount": 0.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.80" } ], "cacheWrite": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00" } ], "other": [] } ] }, { "id": "gpt-5.6-luna", "name": "gpt-5.6-luna", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" }, { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.20" }, { "amount": 1.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.80" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.02" }, { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.04" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10" }, { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.60" }, { "amount": 0.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.90" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.01" }, { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.02" } ], "cacheWrite": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.125" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" } ], "other": [] }, { "name": "flex", "input": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10" }, { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.60" }, { "amount": 0.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.90" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.01" }, { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.02" } ], "cacheWrite": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.125" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" } ], "other": [] }, { "name": "fast", "input": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" }, { "amount": 0.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.80" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.40" }, { "amount": 3.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.60" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.04" }, { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.08" } ], "cacheWrite": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "other": [] } ] }, { "id": "gpt-5.5 (<272K context length)", "name": "gpt-5.5 (<272K context length)", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$45.00" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$22.50" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$22.50" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$75.00" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5.5-pro (<272K context length)", "name": "gpt-5.5-pro (<272K context length)", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$60.00" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$180.00" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$270.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" } ], "output": [ { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$90.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" } ], "output": [ { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$90.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5.4 (<272K context length)", "name": "gpt-5.4 (<272K context length)", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$22.50" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.50" }, { "amount": 11.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$11.25" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.13" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" } ], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.50" }, { "amount": 11.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$11.25" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.13" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" } ], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5.4-mini", "name": "gpt-5.4-mini", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.75" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.50" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.075" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.375" } ], "output": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.25" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0375" } ], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.375" } ], "output": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.25" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0375" } ], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50" } ], "output": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$9.00" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.15" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5.4-nano", "name": "gpt-5.4-nano", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.02" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10" } ], "output": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.625" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.01" } ], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10" } ], "output": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.625" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.01" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5.4-pro (<272K context length)", "name": "gpt-5.4-pro (<272K context length)", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$60.00" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$180.00" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$270.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" } ], "output": [ { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$90.00" }, { "amount": 135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$135.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" } ], "output": [ { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$90.00" }, { "amount": 135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$135.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5.2", "name": "gpt-5.2", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.75" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$14.00" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.175" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.875" } ], "output": [ { "amount": 7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.00" } ], "cacheRead": [ { "amount": 0.0875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0875" } ], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 0.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.875" } ], "output": [ { "amount": 7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.00" } ], "cacheRead": [ { "amount": 0.0875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0875" } ], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.50" } ], "output": [ { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$28.00" } ], "cacheRead": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.35" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5.2-pro", "name": "gpt-5.2-pro", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$21.00" } ], "output": [ { "amount": 168, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$168.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 10.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.50" } ], "output": [ { "amount": 84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$84.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5.1", "name": "gpt-5.1", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.125" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0625" } ], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0625" } ], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$20.00" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5", "name": "gpt-5", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.125" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0625" } ], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0625" } ], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$20.00" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5-mini", "name": "gpt-5-mini", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "cacheRead": [ { "amount": 0.025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.025" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.125" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "cacheRead": [ { "amount": 0.0125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0125" } ], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.125" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "cacheRead": [ { "amount": 0.0125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0125" } ], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 0.45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.45" } ], "output": [ { "amount": 3.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.60" } ], "cacheRead": [ { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.045" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5-nano", "name": "gpt-5-nano", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.05" } ], "output": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" } ], "cacheRead": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.005" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.025" } ], "output": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "cacheRead": [ { "amount": 0.0025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0025" } ], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 0.025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.025" } ], "output": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "cacheRead": [ { "amount": 0.0025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.0025" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5-pro", "name": "gpt-5-pro", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" } ], "output": [ { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$120.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.50" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$60.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-4.1", "name": "gpt-4.1", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$8.00" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.50" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$14.00" } ], "cacheRead": [ { "amount": 0.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.875" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-4.1-mini", "name": "gpt-4.1-mini", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" } ], "output": [ { "amount": 1.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.60" } ], "cacheRead": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "output": [ { "amount": 0.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.80" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.70" } ], "output": [ { "amount": 2.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.80" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.175" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-4.1-nano", "name": "gpt-4.1-nano", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10" } ], "output": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" } ], "cacheRead": [ { "amount": 0.025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.025" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.05" } ], "output": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "output": [ { "amount": 0.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.80" } ], "cacheRead": [ { "amount": 0.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.05" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-4o", "name": "gpt-4o", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 4.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.25" } ], "output": [ { "amount": 17, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$17.00" } ], "cacheRead": [ { "amount": 2.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.125" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-4o-2024-05-13", "name": "gpt-4o-2024-05-13", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.50" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 8.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$8.75" } ], "output": [ { "amount": 26.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$26.25" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-4o-mini", "name": "gpt-4o-mini", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.15" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.60" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.075" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.075" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.30" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.125" } ], "cacheWrite": [], "other": [] } ] }, { "id": "o1", "name": "o1", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$60.00" } ], "cacheRead": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.50" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.50" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "o1-pro", "name": "o1-pro", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 150, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$150.00" } ], "output": [ { "amount": 600, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$600.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$75.00" } ], "output": [ { "amount": 300, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$300.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "o3-pro", "name": "o3-pro", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$20.00" } ], "output": [ { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$80.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00" } ], "output": [ { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$40.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "o3", "name": "o3", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$8.00" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.00" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" } ], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.50" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$14.00" } ], "cacheRead": [ { "amount": 0.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.875" } ], "cacheWrite": [], "other": [] } ] }, { "id": "o4-mini", "name": "o4-mini", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.10" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.40" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.275" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.55" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.20" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.55" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.20" } ], "cacheRead": [ { "amount": 0.138, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.138" } ], "cacheWrite": [], "other": [] }, { "name": "fast", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$8.00" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [], "other": [] } ] }, { "id": "o3-mini", "name": "o3-mini", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.10" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.40" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.55" } ], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.55" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.20" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-4-turbo-2024-04-09", "name": "gpt-4-turbo-2024-04-09", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-4-0613", "name": "gpt-4-0613", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$60.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-3.5-turbo", "name": "gpt-3.5-turbo", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-3.5-turbo-0125", "name": "gpt-3.5-turbo-0125", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.75" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-3.5-turbo-1106", "name": "gpt-3.5-turbo-1106", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-3.5-turbo-instruct", "name": "gpt-3.5-turbo-instruct", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "davinci-002", "name": "davinci-002", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "babbage-002", "name": "babbage-002", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" } ], "output": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "output": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-5.6-cyber", "name": "gpt-5.6-cyber", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$75.00" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" } ], "cacheWrite": [ { "amount": 15.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.625" } ], "other": [] } ] }, { "id": "gpt-5.5-cyber", "name": "gpt-5.5-cyber", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$75.00" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-realtime-2.1", "name": "gpt-realtime-2.1", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$32.00" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$5.00" } ], "output": [ { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$64.00" }, { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$24.00" } ], "cacheRead": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.40" }, { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.50" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-realtime-2.1-mini", "name": "gpt-realtime-2.1-mini", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$10.00" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.60" }, { "amount": 0.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.80" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$20.00" }, { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.40" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.30" }, { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.06" }, { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.08" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-realtime-2", "name": "gpt-realtime-2", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$32.00" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$5.00" } ], "output": [ { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$64.00" }, { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$24.00" } ], "cacheRead": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.40" }, { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.50" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-realtime-1.5", "name": "gpt-realtime-1.5", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$32.00" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$5.00" } ], "output": [ { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$64.00" }, { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$16.00" } ], "cacheRead": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.40" }, { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.50" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-realtime-mini", "name": "gpt-realtime-mini", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$10.00" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.60" }, { "amount": 0.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.80" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$20.00" }, { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.40" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.30" }, { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.06" }, { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.08" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-realtime", "name": "gpt-realtime", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$32.00" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$5.00" } ], "output": [ { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$64.00" }, { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$16.00" } ], "cacheRead": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.40" }, { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.50" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-audio-1.5", "name": "gpt-audio-1.5", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$32.00" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$2.50" } ], "output": [ { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$64.00" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$10.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-audio-mini", "name": "gpt-audio-mini", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$10.00" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.60" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$20.00" }, { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$2.40" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-audio", "name": "gpt-audio", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$32.00" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$2.50" } ], "output": [ { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$64.00" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$10.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-4o-mini-tts", "name": "gpt-4o-mini-tts", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.60" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$12.00" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "tts-1", "name": "tts-1", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$15.00 / 1M characters" } ], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "tts-1-hd", "name": "tts-1-hd", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$30.00 / 1M characters" } ], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-image-2", "name": "gpt-image-2", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$8.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$5.00" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$4.00" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$2.50" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$30.00" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$15.00" } ], "cacheRead": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$2.00" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$1.25" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$1.00" }, { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.625" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-image-1.5", "name": "gpt-image-1.5", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$8.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$5.00" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$4.00" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$2.50" } ], "output": [ { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$32.00" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$10.00" }, { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$16.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$5.00" } ], "cacheRead": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$2.00" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$1.25" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$1.00" }, { "amount": 0.63, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.63" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-image-1-mini", "name": "gpt-image-1-mini", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$2.50" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$2.00" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$1.25" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$1.00" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$8.00" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$4.00" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.25" }, { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.20" }, { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.13" }, { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.10" } ], "cacheWrite": [], "other": [] } ] }, { "id": "gpt-image-1", "name": "gpt-image-1", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$10.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$5.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$5.00" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$2.50" } ], "output": [ { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$40.00" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$20.00" } ], "cacheRead": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$2.50" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$1.25" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$1.25" }, { "amount": 0.63, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.63" } ], "cacheWrite": [], "other": [] } ] }, { "id": "chatgpt-image-latest", "name": "chatgpt-image-latest", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$8.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$5.00" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$4.00" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$2.50" } ], "output": [ { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$32.00" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$10.00" }, { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$16.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$5.00" } ], "cacheRead": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$2.00" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$1.25" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$1.00" }, { "amount": 0.63, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "$0.63" } ], "cacheWrite": [], "other": [] } ] }, { "id": "sora-2", "name": "sora-2", "provider": "openai", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 720, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "720p" }, { "amount": 720, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "720x1280" }, { "amount": 1280, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1280x720" }, { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "$0.10" }, { "amount": 720, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "720p" }, { "amount": 720, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "720x1280" }, { "amount": 1280, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1280x720" }, { "amount": 0.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "$0.05" } ] } ] }, { "id": "sora-2-pro", "name": "sora-2-pro", "provider": "openai", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 720, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "720p" }, { "amount": 720, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "720x1280" }, { "amount": 1280, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1280x720" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "$0.30" }, { "amount": 1024, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1024p" }, { "amount": 1024, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1024x1792" }, { "amount": 1792, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1792x1024" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "$0.50" }, { "amount": 1080, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1080p" }, { "amount": 1080, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1080x1920" }, { "amount": 1920, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1920x1080" }, { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "$0.70" }, { "amount": 720, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "720p" }, { "amount": 720, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "720x1280" }, { "amount": 1280, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1280x720" }, { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "$0.15" }, { "amount": 1024, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1024p" }, { "amount": 1024, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1024x1792" }, { "amount": 1792, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1792x1024" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "$0.25" }, { "amount": 1080, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1080p" }, { "amount": 1080, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1080x1920" }, { "amount": 1920, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "1920x1080" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "$0.35" } ] } ] }, { "id": "gpt-realtime-translate", "name": "gpt-realtime-translate", "provider": "openai", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.034, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.034 / minute" } ] } ] }, { "id": "gpt-live-transcribe", "name": "gpt-live-transcribe", "provider": "openai", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.017, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.017 / minute" } ] } ] }, { "id": "gpt-realtime-whisper", "name": "gpt-realtime-whisper", "provider": "openai", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.017, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.017 / minute" } ] } ] }, { "id": "gpt-transcribe", "name": "gpt-transcribe", "provider": "openai", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.0045 / minute" } ] } ] }, { "id": "gpt-4o-transcribe", "name": "gpt-4o-transcribe", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$2.50" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$10.00" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.006, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.006 / minute" } ] } ] }, { "id": "gpt-4o-mini-transcribe", "name": "gpt-4o-mini-transcribe", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$1.25" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$5.00" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.003 / minute" } ] } ] }, { "id": "gpt-4o-transcribe-diarize", "name": "gpt-4o-transcribe-diarize", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$2.50" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$10.00" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.006, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.006 / minute" } ] } ] }, { "id": "Whisper", "name": "Whisper", "provider": "openai", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.006, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "$0.006 / minute" } ] } ] }, { "id": "Web search", "name": "Web search", "provider": "openai", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00 / 1k calls + Search content tokens billed at model rates." }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00 / 1k calls + Search content tokens billed at model rates." }, { "amount": -5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "Web search preview (reasoning models, including gpt-5, o-series)" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00 / 1k calls + Search content tokens billed at model rates." }, { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25.00 / 1k calls + Search content tokens are free." } ] } ] }, { "id": "Containers", "name": "Containers", "provider": "openai", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1 GB $0.03, 4 GB $0.12, 16 GB $0.48, 64 GB $1.92 per 20-minute session per container." } ] } ] }, { "id": "File search", "name": "File search", "provider": "openai", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10 / GB per day (1 GB free)" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50 / 1k calls" } ] } ] }, { "id": "Agent Kit", "name": "Agent Kit", "provider": "openai", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10 / GB-day after 1 GB free per account per month" } ] } ] }, { "id": "ChatGPT", "name": "ChatGPT", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30.00" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [], "other": [] } ] }, { "id": "Codex", "name": "Codex", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.75" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.50" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$14.00" }, { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$28.00" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.175" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.35" } ], "cacheWrite": [], "other": [ { "amount": -5.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "gpt-5.3-codex" }, { "amount": -5.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "gpt-5.3-codex" } ] } ] }, { "id": "Search", "name": "Search", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10.00" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.125" } ], "cacheWrite": [], "other": [ { "amount": -5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "gpt-5-search-api" } ] } ] }, { "id": "Embedding", "name": "Embedding", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "$0.02" }, { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "$0.13" }, { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "$0.10" } ], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": -3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "text-embedding-3-small" }, { "amount": -3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "text-embedding-3-large" }, { "amount": -2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "text-embedding-ada-002" } ] } ] }, { "id": "o4-mini-2025-04-16", "name": "o4-mini-2025-04-16", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.00" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$16.00" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$8.00" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [], "other": [ { "amount": 100, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$100.00 / hour" }, { "amount": 100, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$100.00 / hour" } ] } ] }, { "id": "o4-mini-2025-04-16 (data sharing)", "name": "o4-mini-2025-04-16 (data sharing)", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.00" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.00" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$8.00" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4.00" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.25" } ], "cacheWrite": [], "other": [ { "amount": 100, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$100.00 / hour" }, { "amount": 100, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$100.00 / hour" } ] } ] }, { "id": "gpt-4.1-2025-04-14", "name": "gpt-4.1-2025-04-14", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.00" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.00" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.00" } ], "cacheRead": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.75" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50" } ], "cacheWrite": [], "other": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25.00" }, { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25.00" } ] } ] }, { "id": "gpt-4.1-mini-2025-04-14", "name": "gpt-4.1-mini-2025-04-14", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.80" }, { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" } ], "output": [ { "amount": 3.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.20" }, { "amount": 1.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.60" } ], "cacheRead": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" }, { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10" } ], "cacheWrite": [], "other": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5.00" } ] } ] }, { "id": "gpt-4.1-nano-2025-04-14", "name": "gpt-4.1-nano-2025-04-14", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20" }, { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10" } ], "output": [ { "amount": 0.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.80" }, { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" } ], "cacheRead": [ { "amount": 0.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.05" }, { "amount": 0.025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.025" } ], "cacheWrite": [], "other": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50" } ] } ] }, { "id": "gpt-4o-2024-08-06", "name": "gpt-4o-2024-08-06", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.75" }, { "amount": 2.225, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.225" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15.00" }, { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50" } ], "cacheRead": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.875" }, { "amount": 0.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.90" } ], "cacheWrite": [], "other": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25.00" }, { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25.00" } ] } ] }, { "id": "gpt-4o-mini-2024-07-18", "name": "gpt-4o-mini-2024-07-18", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.30" }, { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.15" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.20" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.60" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.15" }, { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.075" } ], "cacheWrite": [], "other": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.00" }, { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.00" } ] } ] }, { "id": "gpt-3.5-turbo (legacy)", "name": "gpt-3.5-turbo (legacy)", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.00" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.00" }, { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.00" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$8.00" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$8.00" } ] } ] }, { "id": "davinci-002 (legacy)", "name": "davinci-002 (legacy)", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.00" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.00" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.00" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.00" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.00" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.00" } ] } ] }, { "id": "babbage-002 (legacy)", "name": "babbage-002 (legacy)", "provider": "openai", "tiers": [ { "name": "standard", "input": [ { "amount": 1.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.60" }, { "amount": 0.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.80" } ], "output": [ { "amount": 1.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.60" }, { "amount": 0.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.90" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" }, { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40" } ] } ] } ] }, { "provider": "anthropic", "source": { "url": "https://platform.claude.com/docs/en/about-claude/pricing.md", "fetchedAt": "2026-08-19T00:56:01.107Z" }, "models": [ { "id": "claude-fable-5", "name": "Claude Fable 5", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10 / MTok" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$50 / MTok" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1 / MTok" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50 / MTok" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$20 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5 / MTok" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-mythos-5-limited-availability", "name": "Claude Mythos 5 (limited availability)", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10 / MTok" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$50 / MTok" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1 / MTok" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50 / MTok" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$20 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5 / MTok" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-opus-5", "name": "Claude Opus 5", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5 / MTok" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25 / MTok" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50 / MTok" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.25 / MTok" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50 / MTok" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-opus-4-8", "name": "Claude Opus 4.8", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5 / MTok" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25 / MTok" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50 / MTok" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.25 / MTok" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50 / MTok" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-opus-4-7", "name": "Claude Opus 4.7", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5 / MTok" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25 / MTok" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50 / MTok" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.25 / MTok" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50 / MTok" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-opus-4-6", "name": "Claude Opus 4.6", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5 / MTok" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25 / MTok" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50 / MTok" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.25 / MTok" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50 / MTok" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-opus-4-5", "name": "Claude Opus 4.5", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5 / MTok" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$25 / MTok" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50 / MTok" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6.25 / MTok" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50 / MTok" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$12.50 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-opus-4-1-retired-except-on-bedrock-and-google-cloud", "name": "Claude Opus 4.1 (retired, except on Bedrock and Google Cloud)", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15 / MTok" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$75 / MTok" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50 / MTok" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$18.75 / MTok" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.50 / MTok" } ], "output": [ { "amount": 37.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$37.50 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-opus-4-retired-except-on-google-cloud", "name": "Claude Opus 4 (retired, except on Google Cloud)", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15 / MTok" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$75 / MTok" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50 / MTok" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$18.75 / MTok" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$30 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.50 / MTok" } ], "output": [ { "amount": 37.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$37.50 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-sonnet-5", "name": "Claude Sonnet 5", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2 / MTok" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$10 / MTok" } ], "cacheRead": [ { "amount": 0.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.20 / MTok" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50 / MTok" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1 / MTok" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-sonnet-4-6", "name": "Claude Sonnet 4.6", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3 / MTok" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15 / MTok" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.30 / MTok" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.75 / MTok" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50 / MTok" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.50 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-sonnet-4-5", "name": "Claude Sonnet 4.5", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3 / MTok" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15 / MTok" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.30 / MTok" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.75 / MTok" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50 / MTok" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.50 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-sonnet-4-retired-except-on-bedrock-and-google-cloud", "name": "Claude Sonnet 4 (retired, except on Bedrock and Google Cloud)", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3 / MTok" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$15 / MTok" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.30 / MTok" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.75 / MTok" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$6 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.50 / MTok" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$7.50 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-haiku-4-5", "name": "Claude Haiku 4.5", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1 / MTok" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$5 / MTok" } ], "cacheRead": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.10 / MTok" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.25 / MTok" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.50 / MTok" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2.50 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "claude-haiku-3-5-retired-except-on-bedrock-and-google-cloud", "name": "Claude Haiku 3.5 (retired, except on Bedrock and Google Cloud)", "provider": "anthropic", "tiers": [ { "name": "standard", "input": [ { "amount": 0.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.80 / MTok" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$4 / MTok" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.08 / MTok" } ], "cacheWrite": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1 / MTok" }, { "amount": 1.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$1.60 / MTok" } ], "other": [] }, { "name": "batch", "input": [ { "amount": 0.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.40 / MTok" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$2 / MTok" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] } ] }, { "provider": "google", "source": { "url": "https://ai.google.dev/gemini-api/docs/pricing.md.txt", "fetchedAt": "2026-08-19T00:56:01.674Z" }, "models": [ { "id": "gemini-3.7-flash", "name": "Gemini 3.7 Flash", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.75, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.75 through December 31, 2026. $1.50 starting January 1, 2027." } ], "output": [ { "amount": 3.75, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$3.75 through December 31, 2026. $7.50 starting January 1, 2027." } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.075 through December 31, 2026. $0.15 starting January 1, 2027. $0.50 / 1,000,000 tokens per hour (storage price) through December 31, 2026. $1.00 / 1,000,000 tokens per hour (storage price) starting January 1, 2027." } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "batch", "input": [ { "amount": 0.375, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.375 through December 31, 2026. $0.75 starting January 1, 2027." } ], "output": [ { "amount": 1.875, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.875 through December 31, 2026. $3.75 starting January 1, 2027." } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.0375 through December 31, 2026. $0.075 starting January 1, 2027. $0.50 / 1,000,000 tokens per hour (storage price) through December 31, 2026. $1.00 / 1,000,000 tokens per hour (storage price) starting January 1, 2027." } ], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 0.375, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.375 through December 31, 2026. $0.75 starting January 1, 2027." } ], "output": [ { "amount": 1.875, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.875 through December 31, 2026. $3.75 starting January 1, 2027." } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.0375 through December 31, 2026. $0.075 starting January 1, 2027. $0.50 / 1,000,000 tokens per hour (storage price) through December 31, 2026. $1.00 / 1,000,000 tokens per hour (storage price) starting January 1, 2027." } ], "cacheWrite": [], "other": [] }, { "name": "priority", "input": [ { "amount": 1.35, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.35 through December 31, 2026. $2.70 starting January 1, 2027." } ], "output": [ { "amount": 6.75, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$6.75 through December 31, 2026. $13.50 starting January 1, 2027." } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.135 through December 31, 2026. $0.27 starting January 1, 2027. $0.50 / 1,000,000 tokens per hour (storage price) through December 31, 2026. $1.00 / 1,000,000 tokens per hour (storage price) starting January 1, 2027." } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] } ] }, { "id": "gemini-3.6-flash", "name": "Gemini 3.6 Flash", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.75, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.75 through December 31, 2026. $1.50 starting January 1, 2027." } ], "output": [ { "amount": 3.75, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$3.75 through December 31, 2026. $7.50 starting January 1, 2027." } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.075 through December 31, 2026. $0.15 starting January 1, 2027. $0.50 / 1,000,000 tokens per hour (storage price) through December 31, 2026. $1.00 / 1,000,000 tokens per hour (storage price) starting January 1, 2027." } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "batch", "input": [ { "amount": 0.375, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.375 through December 31, 2026. $0.75 starting January 1, 2027." } ], "output": [ { "amount": 1.875, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.875 through December 31, 2026. $3.75 starting January 1, 2027." } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.0375 through December 31, 2026. $0.075 starting January 1, 2027. $0.50 / 1,000,000 tokens per hour (storage price) through December 31, 2026. $1.00 / 1,000,000 tokens per hour (storage price) starting January 1, 2027." } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 requests per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "flex", "input": [ { "amount": 0.375, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.375 through December 31, 2026. $0.75 starting January 1, 2027." } ], "output": [ { "amount": 1.875, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.875 through December 31, 2026. $3.75 starting January 1, 2027." } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.0375 through December 31, 2026. $0.075 starting January 1, 2027. $0.50 / 1,000,000 tokens per hour (storage price) through December 31, 2026. $1.00 / 1,000,000 tokens per hour (storage price) starting January 1, 2027." } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 requests per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "priority", "input": [ { "amount": 1.35, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.35 through December 31, 2026. $2.70 starting January 1, 2027." } ], "output": [ { "amount": 6.75, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$6.75 through December 31, 2026. $13.50 starting January 1, 2027." } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.135 through December 31, 2026. $0.27 starting January 1, 2027. $0.50 / 1,000,000 tokens per hour (storage price) through December 31, 2026. $1.00 / 1,000,000 tokens per hour (storage price) starting January 1, 2027." } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] } ] }, { "id": "gemini-3.5-flash", "name": "Gemini 3.5 Flash", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 1.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.50" } ], "output": [ { "amount": 9, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$9.00" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.15 $1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "batch", "input": [ { "amount": 0.75, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.75" } ], "output": [ { "amount": 4.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$4.50" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.075 $1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 requests per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "flex", "input": [ { "amount": 0.75, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.75" } ], "output": [ { "amount": 4.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$4.50" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.08 $1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 requests per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "priority", "input": [ { "amount": 2.7, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$2.70" } ], "output": [ { "amount": 16.2, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$16.20" } ], "cacheRead": [ { "amount": 0.27, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.27 $1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] } ] }, { "id": "gemini-3.5-live-translate-preview", "name": "Gemini 3.5 Live Translate", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 3.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$3.50 or $0.0053/min^^ (audio)" } ], "output": [ { "amount": 21, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$21.00 or $0.0315/min^^ (audio)" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gemini-3.5-flash-lite", "name": "Gemini 3.5 Flash-Lite", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.3, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.30 (text / image / video / audio)" } ], "output": [ { "amount": 2.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$2.50" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.03 $1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 requests per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "batch", "input": [ { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.15 (text / image / video / audio)" } ], "output": [ { "amount": 1.25, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.25" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.02 $1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "flex", "input": [ { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.15 (text / image / video / audio)" } ], "output": [ { "amount": 1.25, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.25" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.02 $1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "priority", "input": [ { "amount": 0.54, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.54 (text / image / video / audio)" } ], "output": [ { "amount": 4.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$4.50" } ], "cacheRead": [ { "amount": 0.05, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.05 $1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] } ] }, { "id": "gemini-3.1-flash-lite", "name": "Gemini 3.1 Flash-Lite", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.25, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.25 (text / image / video)" }, { "amount": 0.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.50 (audio)" } ], "output": [ { "amount": 1.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.50" } ], "cacheRead": [ { "amount": 0.025, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.025 (text / image / video)" }, { "amount": 0.05, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.05 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "batch", "input": [ { "amount": 0.125, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.125 (text / image / video)" }, { "amount": 0.25, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.25 (audio)" } ], "output": [ { "amount": 0.75, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.75" } ], "cacheRead": [ { "amount": 0.0125, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.0125 (text / image / video)" }, { "amount": 0.025, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.025 (audio)" }, { "amount": 0.5, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.50 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "flex", "input": [ { "amount": 0.125, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.125 (text / image / video)" }, { "amount": 0.25, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.25 (audio)" } ], "output": [ { "amount": 0.75, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.75" } ], "cacheRead": [ { "amount": 0.0125, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.0125 (text / image / video)" }, { "amount": 0.025, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.025 (audio)" }, { "amount": 0.5, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.50 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "priority", "input": [ { "amount": 0.45, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.45 (text / image / video)" }, { "amount": 0.9, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.90 (audio)" } ], "output": [ { "amount": 2.7, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$2.70" } ], "cacheRead": [ { "amount": 0.045, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.045 (text / image / video)" }, { "amount": 0.09, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.09 (audio)" }, { "amount": 1.8, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.80 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] } ] }, { "id": "gemini-omni-flash-preview", "name": "Gemini Omni Flash Preview", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 1.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.50 (text / image / video / audio)" } ], "output": [ { "amount": 9, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$9.00 (text)" }, { "amount": 17.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$17.50 (video)^^" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gemini-3.1-pro-preview", "name": "Gemini 3.1 Pro Preview", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 2, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$2.00, prompts <= 200k tokens $4.00, prompts > 200k tokens" } ], "output": [ { "amount": 12, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$12.00, prompts <= 200k tokens $18.00, prompts > 200k" } ], "cacheRead": [ { "amount": 0.2, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.20, prompts <= 200k tokens $0.40, prompts > 200k $4.50 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "batch", "input": [ { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.00, prompts <= 200k tokens $2.00, prompts > 200k tokens" } ], "output": [ { "amount": 6, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$6.00, prompts <= 200k tokens $9.00, prompts > 200k" } ], "cacheRead": [ { "amount": 0.2, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "Same as Standard $0.20, prompts <= 200k tokens $0.40, prompts > 200k $4.50 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 requests per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "flex", "input": [ { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.00, prompts <= 200k tokens $2.00, prompts > 200k tokens" } ], "output": [ { "amount": 6, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$6.00, prompts <= 200k tokens $9.00, prompts > 200k" } ], "cacheRead": [ { "amount": 0.2, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "Same as Standard $0.20, prompts <= 200k tokens $0.40, prompts > 200k $4.50 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 requests per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "priority", "input": [ { "amount": 3.6, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$3.60, prompts <= 200k tokens $7.20, prompts > 200k tokens" } ], "output": [ { "amount": 21.6, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$21.60, prompts <= 200k tokens $32.40, prompts > 200k" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.36, prompts <= 200k tokens $0.72, prompts > 200k $8.10 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] } ] }, { "id": "gemini-3.1-flash-live-preview", "name": "Gemini 3.1 Flash Live Preview", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.75, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$0.75 (text)" }, { "amount": 3, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$3.00 or $0.005/min (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.00 or $0.002/min (image/video)" } ], "output": [ { "amount": 4.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$4.50 (text)" }, { "amount": 12, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$12.00 or $0.018/min (audio)" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." } ] } ] }, { "id": "gemini-3.1-flash-image", "name": "Gemini 3.1 Flash Image (Nano Banana 2) 🍌", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.5, "currency": "USD", "pricingType": "image", "units": 1, "raw": "$0.50 (text/image)" } ], "output": [ { "amount": 3, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$3 (text and thinking)" }, { "amount": 60, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$60.00 (images) Equivalent to $0.045 per 0.5K image^^ $0.067 per 1K image^^, $0.101 per 2K image^^, and $0.151 per 4K image^^." } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests for text and image-based grounding." } ] }, { "name": "batch", "input": [ { "amount": 0.25, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$0.25 (text, image)" } ], "output": [ { "amount": 1.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$1.50 (text and thinking)" }, { "amount": 30, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$30.00 (images) Equivalent to $0.022 per 0.5K image^^ $0.034 per 1K image^^, $0.050 per 2K image^^, and $0.076 per 4K image^^." } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gemini-3.1-flash-lite-image", "name": "Gemini 3.1 Flash Lite Image (Nano Banana 2 Lite) 🍌", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.25, "currency": "USD", "pricingType": "image", "units": 1, "raw": "$0.25 (text/image/video)" } ], "output": [ { "amount": 1.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$1.50 (text and thinking)" }, { "amount": 30, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$30.00 (images) Equivalent to $0.0336 per 1K resolution image^^" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.125, "currency": "USD", "pricingType": "image", "units": 1, "raw": "$0.125 (text/image/video)" } ], "output": [ { "amount": 0.75, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$0.75 (text and thinking)" }, { "amount": 15, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$15.00 (images) Equivalent to $0.0168 per 1K resolution image^^" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gemini-3.1-flash-tts-preview", "name": "Gemini 3.1 Flash TTS Preview", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$1.00 (text)" } ], "output": [ { "amount": 20, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$20.00 (audio)" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$0.50 (text)" } ], "output": [ { "amount": 10, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$10.00 (audio)" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gemini-3-flash-preview", "name": "Gemini 3 Flash Preview", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.50 (text / image / video)" }, { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$1.00 (audio)" } ], "output": [ { "amount": 3, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$3.00" } ], "cacheRead": [ { "amount": 0.05, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.05 (text / image / video)" }, { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.10 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "batch", "input": [ { "amount": 0.25, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.25 (text / image / video)" }, { "amount": 0.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.50 (audio)" } ], "output": [ { "amount": 1.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.50" } ], "cacheRead": [ { "amount": 0.05, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "Same as Standard $0.05 (text / image / video)" }, { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.10 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 requests per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "flex", "input": [ { "amount": 0.25, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.25 (text / image / video)" }, { "amount": 0.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.50 (audio)" } ], "output": [ { "amount": 1.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.50" } ], "cacheRead": [ { "amount": 0.05, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "Same as Standard $0.05 (text / image / video)" }, { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.10 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 requests per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] }, { "name": "priority", "input": [ { "amount": 0.9, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.90 (text / image / video)" }, { "amount": 1.8, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$1.80 (audio)" } ], "output": [ { "amount": 5.4, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$5.40" } ], "cacheRead": [ { "amount": 0.09, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.09 (text / image / video)" }, { "amount": 0.18, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.18 (audio)" }, { "amount": 1.8, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.80 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." }, { "amount": 5000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "5,000 prompts per month (free, shared across Gemini 3), then $14 / 1,000 search queries" } ] } ] }, { "id": "gemini-3-pro-image", "name": "Gemini 3 Pro Image (Nano Banana Pro) 🍌", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 2, "currency": "USD", "pricingType": "image", "units": 1, "raw": "$2.00 (text/image), equivalent to $0.0011 per image^^" } ], "output": [ { "amount": 12, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$12.00 (text and thinking)" }, { "amount": 120, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$120.00 (images) Equivalent to $0.134 per 1K/2K image^^ and $0.24 per 4K image^^" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." } ] }, { "name": "batch", "input": [ { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$1.00 (text), $0.0006 (image)^^" } ], "output": [ { "amount": 6, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$6.00 (text and thinking)" }, { "amount": 0.067, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$0.067 per 1K/2K image^^ $0.12 per 4K image^^" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$1.00 (text), $0.0006 (image)^^" } ], "output": [ { "amount": 6, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$6.00 (text and thinking)" }, { "amount": 0.067, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$0.067 per 1K/2K image^^ $0.12 per 4K image^^" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "priority", "input": [ { "amount": 3.6, "currency": "USD", "pricingType": "image", "units": 1, "raw": "$3.60 (text/image)" } ], "output": [ { "amount": 21.6, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$21.60 (text and thinking)" }, { "amount": 216, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$216.00 (images)" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." } ] } ] }, { "id": "gemini-2.5-pro", "name": "Gemini 2.5 Pro", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 1.25, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.25, prompts <= 200k tokens $2.50, prompts > 200k tokens" } ], "output": [ { "amount": 10, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$10.00, prompts <= 200k tokens $15.00, prompts > 200k" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.125, prompts <= 200k tokens $0.25, prompts > 200k $4.50 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free), then $35 / 1,000 grounded prompts" }, { "amount": 10000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "10,000 RPD (free), then $25 / 1,000 grounded prompts" } ] }, { "name": "batch", "input": [ { "amount": 0.625, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.625, prompts <= 200k tokens $1.25, prompts > 200k tokens" } ], "output": [ { "amount": 5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$5.00, prompts <= 200k tokens $7.50, prompts > 200k" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.125, prompts <= 200k tokens $0.25, prompts > 200k $4.50 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free), then $35 / 1,000 grounded prompts" } ] }, { "name": "flex", "input": [ { "amount": 0.625, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.625, prompts <= 200k tokens $1.25, prompts > 200k tokens" } ], "output": [ { "amount": 5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$5.00, prompts <= 200k tokens $7.50, prompts > 200k" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.125, prompts <= 200k tokens $0.25, prompts > 200k $4.50 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free), then $35 / 1,000 grounded prompts" } ] }, { "name": "priority", "input": [ { "amount": 2.25, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$2.25, prompts <= 200k tokens $4.50, prompts > 200k tokens" } ], "output": [ { "amount": 18, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$18.00, prompts <= 200k tokens $27.00, prompts > 200k" } ], "cacheRead": [ { "amount": 0.225, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.225, prompts <= 200k tokens $0.45, prompts > 200k $8.10 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free), then $35 / 1,000 grounded prompts" }, { "amount": 10000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "10,000 RPD (free), then $25 / 1,000 grounded prompts" } ] } ] }, { "id": "gemini-2.5-flash", "name": "Gemini 2.5 Flash", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.3, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.30 (text / image / video)" }, { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$1.00 (audio)" } ], "output": [ { "amount": 2.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$2.50" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.03 (text / image / video)" }, { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.1 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free, limit shared with Flash-Lite RPD), then $35 / 1,000 grounded prompts" }, { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free), then $25 / 1,000 grounded prompts" } ] }, { "name": "batch", "input": [ { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.15 (text / image / video)" }, { "amount": 0.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.50 (audio)" } ], "output": [ { "amount": 1.25, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.25" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.03 (text / image / video)" }, { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.1 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free, limit shared with Flash-Lite RPD), then $35 / 1,000 grounded prompts" } ] }, { "name": "flex", "input": [ { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.15 (text / image / video)" }, { "amount": 0.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.50 (audio)" } ], "output": [ { "amount": 1.25, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.25" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.03 (text / image / video)" }, { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.1 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free, limit shared with Flash-Lite RPD), then $35 / 1,000 grounded prompts" } ] }, { "name": "priority", "input": [ { "amount": 0.54, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.54 (text / image / video)" }, { "amount": 1.8, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$1.80 (audio)" } ], "output": [ { "amount": 4.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$4.50" } ], "cacheRead": [ { "amount": 0.054, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.054 (text / image / video)" }, { "amount": 0.18, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.18 (audio)" }, { "amount": 1.8, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.80 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free, limit shared with Flash-Lite RPD), then $35 / 1,000 grounded prompts" }, { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free), then $25 / 1,000 grounded prompts" } ] } ] }, { "id": "gemini-2.5-flash-lite", "name": "Gemini 2.5 Flash-Lite", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.10 (text / image / video)" }, { "amount": 0.3, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.30 (audio)" } ], "output": [ { "amount": 0.4, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.40" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.01 (text / image / video)" }, { "amount": 0.03, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.03 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free, limit shared with Flash RPD), then $35 / 1,000 grounded prompts" }, { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free), then $25 / 1,000 grounded prompts" } ] }, { "name": "batch", "input": [ { "amount": 0.05, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.05 (text / image / video)" }, { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.15 (audio)" } ], "output": [ { "amount": 0.2, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.20" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.01 (text / image / video)" }, { "amount": 0.03, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.03 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free, limit shared with Flash RPD), then $35 / 1,000 grounded prompts" } ] }, { "name": "flex", "input": [ { "amount": 0.05, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.05 (text / image / video)" }, { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.15 (audio)" } ], "output": [ { "amount": 0.2, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.20" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.01 (text / image / video)" }, { "amount": 0.03, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.03 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free, limit shared with Flash RPD), then $35 / 1,000 grounded prompts" } ] }, { "name": "priority", "input": [ { "amount": 0.18, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.18 (text / image / video)" }, { "amount": 0.54, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.54 (audio)" } ], "output": [ { "amount": 0.72, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.72" } ], "cacheRead": [ { "amount": 0.018, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.018 (text / image / video)" }, { "amount": 0.054, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.054 (audio)" }, { "amount": 1.8, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.80 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free, limit shared with Flash RPD), then $35 / 1,000 grounded prompts" }, { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free), then $25 / 1,000 grounded prompts" } ] } ] }, { "id": "gemini-2.5-flash-lite-preview-09-2025", "name": "Gemini 2.5 Flash-Lite Preview", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.10 (text / image / video)" }, { "amount": 0.3, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.30 (audio)" } ], "output": [ { "amount": 0.4, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.40" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.01 (text / image / video)" }, { "amount": 0.03, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.03 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free, limit shared with Flash RPD), then $35 / 1,000 grounded prompts" } ] }, { "name": "batch", "input": [ { "amount": 0.05, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.05 (text / image / video)" }, { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.15 (audio)" } ], "output": [ { "amount": 0.2, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.20" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.01 (text / image / video)" }, { "amount": 0.03, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.03 (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free, limit shared with Flash RPD), then $35 / 1,000 grounded prompts" } ] } ] }, { "id": "gemini-2.5-flash-native-audio-preview-12-2025", "name": "Gemini 2.5 Flash Native Audio (Live API)", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$0.50 (text)" }, { "amount": 3, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$3.00 (audio / video)" } ], "output": [ { "amount": 2, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$2.00 (text)" }, { "amount": 12, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$12.00 (audio)" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gemini-2.5-flash-image", "name": "Gemini 2.5 Flash Image (Nano Banana) 🍌", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.3, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$0.30 (text / image)" } ], "output": [ { "amount": 0.039, "currency": "USD", "pricingType": "image", "units": 1, "modality": "image", "raw": "$0.039 per image" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$0.15 (text / image)" } ], "output": [ { "amount": 0.0195, "currency": "USD", "pricingType": "image", "units": 1, "modality": "image", "raw": "$0.0195 per image" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "flex", "input": [ { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$0.15 (text / image)" } ], "output": [ { "amount": 0.0195, "currency": "USD", "pricingType": "image", "units": 1, "modality": "image", "raw": "$0.0195 per image" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "priority", "input": [ { "amount": 0.54, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$0.54 (text / image)" } ], "output": [ { "amount": 0.0702, "currency": "USD", "pricingType": "image", "units": 1, "modality": "image", "raw": "$0.0702 per image" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gemini-2.5-flash-preview-tts", "name": "Gemini 2.5 Flash Preview TTS", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$0.50 (text)" } ], "output": [ { "amount": 10, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$10.00 (audio)" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.25, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$0.25 (text)" } ], "output": [ { "amount": 5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$5.00 (audio)" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gemini-2.5-pro-preview-tts", "name": "Gemini 2.5 Pro Preview TTS", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$1.00 (text)" } ], "output": [ { "amount": 20, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$20.00 (audio)" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "text", "raw": "$0.50 (text)" } ], "output": [ { "amount": 10, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$10.00 (audio)" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gemini-2.0-flash", "name": "Gemini 2.0 Flash", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.10 (text / image / video)" }, { "amount": 0.7, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.70 (audio)" } ], "output": [ { "amount": 0.4, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.40" } ], "cacheRead": [ { "amount": 0.025, "currency": "USD", "pricingType": "image", "units": 1, "raw": "$0.025 / 1,000,000 tokens (text/image/video)" }, { "amount": 0.175, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.175 / 1,000,000 tokens (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free), then $35 / 1,000 grounded prompts" }, { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free), then $25 / 1,000 grounded prompts" } ] }, { "name": "batch", "input": [ { "amount": 0.05, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.05 (text / image / video)" }, { "amount": 0.35, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.35 (audio)" } ], "output": [ { "amount": 0.2, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.20" } ], "cacheRead": [ { "amount": 0.025, "currency": "USD", "pricingType": "image", "units": 1, "raw": "$0.025 / 1,000,000 tokens (text/image/video)" }, { "amount": 0.175, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.175 / 1,000,000 tokens (audio)" }, { "amount": 1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$1.00 / 1,000,000 tokens per hour" } ], "cacheWrite": [], "other": [ { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD (free), then $35 / 1,000 grounded prompts" } ] } ] }, { "id": "gemini-2.0-flash-lite", "name": "Gemini 2.0 Flash-Lite", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.075, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.075" } ], "output": [ { "amount": 0.3, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.30" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.0375, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.0375" } ], "output": [ { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.15" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "imagen-4.0-generate-001", "name": "Imagen 4", "provider": "google", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$0.02" }, { "amount": 0.04, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$0.04" }, { "amount": 0.06, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "image", "raw": "$0.06" } ] } ] }, { "id": "veo-3.1-generate-preview", "name": "Veo 3.1", "provider": "google", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.4, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.40 (720p and 1080p)" }, { "amount": 0.6, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.60 (4k)" }, { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.10 (720p)" }, { "amount": 0.12, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.12 (1080p)" }, { "amount": 0.3, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.30 (4k)" }, { "amount": 0.05, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.05 (720p)" }, { "amount": 0.08, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.08 (1080p) (4k output not supported)" } ] } ] }, { "id": "veo-3.0-generate-001", "name": "Veo 3", "provider": "google", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.4, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.40" }, { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.10 (720p)" }, { "amount": 0.12, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.12 (1080p)" }, { "amount": 0.3, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.30 (4k)" } ] } ] }, { "id": "veo-2.0-generate-001", "name": "Veo 2", "provider": "google", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.35, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "video", "raw": "$0.35" } ] } ] }, { "id": "lyria-3-clip-preview", "name": "Lyria 3", "provider": "google", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.04 per song" }, { "amount": 0.08, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$0.08 per song" } ] } ] }, { "id": "gemini-embedding-2", "name": "Gemini Embedding 2", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.2, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "embedding", "raw": "$0.20" }, { "amount": 0.45, "currency": "USD", "pricingType": "image", "units": 1, "modality": "image", "raw": "$0.45 ($0.00012 per image)" }, { "amount": 6.5, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "embedding", "raw": "$6.50 ($0.00016 per second)" }, { "amount": 12, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "embedding", "raw": "$12.00 ($0.00079 per frame)" } ], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "embedding", "raw": "$0.10" }, { "amount": 0.225, "currency": "USD", "pricingType": "image", "units": 1, "modality": "image", "raw": "$0.225 ($0.00006 per image)" }, { "amount": 3.25, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "embedding", "raw": "$3.25 ($0.00008 per second)" }, { "amount": 6, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "embedding", "raw": "$6.00 ($0.000395 per frame)" } ], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gemini-embedding-001", "name": "Gemini Embedding", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "embedding", "raw": "$0.15" } ], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch", "input": [ { "amount": 0.075, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "embedding", "raw": "$0.075" } ], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] } ] }, { "id": "gemini-robotics-er-2-preview", "name": "Gemini Robotics ER 2 Preview", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 2, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$2.00 (text / image / video / audio)" } ], "output": [ { "amount": 10, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$10.00" } ], "cacheRead": [ { "amount": 0.2, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.20 $1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." } ] }, { "name": "batch", "input": [ { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.00 (text / image / video / audio)" } ], "output": [ { "amount": 5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$5.00" } ], "cacheRead": [ { "amount": 0.1, "currency": "USD", "pricingType": "hour", "units": 1, "raw": "$0.10 $1.00 / 1,000,000 tokens per hour (storage price)" } ], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." } ] } ] }, { "id": "gemini-robotics-er-2-streaming-preview", "name": "Gemini Robotics ER 2 Streaming Preview", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 2, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$2.00 (text / image / video / audio)" } ], "output": [ { "amount": 10, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$10.00" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." } ] } ] }, { "id": "gemini-robotics-er-1.6-preview", "name": "Gemini Robotics ER 1.6 Preview", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.00 (text / image / video)" }, { "amount": 2, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$2.00 (audio)" } ], "output": [ { "amount": 5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$5.00" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." } ] }, { "name": "batch", "input": [ { "amount": 0.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$0.50 (text / image / video)" }, { "amount": 1, "currency": "USD", "pricingType": "token", "units": 1000000, "modality": "audio", "raw": "$1.00 (audio)" } ], "output": [ { "amount": 2.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$2.50" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 5000, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "5,000 free search requests per month (shared across all Gemini 3.x models), then $14 per 1,000 requests." } ] } ] }, { "id": "gemini-2.5-computer-use-preview-10-2025", "name": "Gemini 2.5 Computer Use Preview", "provider": "google", "tiers": [ { "name": "standard", "input": [ { "amount": 1.25, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$1.25, prompts <= 200k tokens $2.50, prompts > 200k token" } ], "output": [ { "amount": 10, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "$10.00, prompts <= 200k tokens $15.00, prompts > 200k" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "pricingType": "request", "units": 1000, "raw": "Gemini 2.5 models: 1,500 RPD free (limit shared for Flash and Flash-Lite). Then $35 / 1,000 grounded prompts Gemini 3 models: 5,000 free search requests per month (shared across all Gemini models), then $14 per 1,000 requests." }, { "amount": 1500, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "1,500 RPD free (limit shared for Flash and Flash-Lite)" }, { "amount": 10000, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "10,000 RPD free for Pro. Then $25 / 1,000 grounded prompts" }, { "amount": 3.5, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "Charged as regular tokens per model pricing (e.g., standard Gemini 3.5 Flash pricing). See the Gemini 2.5 Computer Use Preview pricing table for legacy model rates." }, { "amount": 0.15, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "Charged for embeddings at $0.15 / 1M tokens. Retrieved document tokens charged as regular tokens per model pricing." }, { "amount": 3.1, "currency": "USD", "pricingType": "token", "units": 1000000, "raw": "Same as Gemini 3.1 Pro Preview pricing" } ] } ] } ] }, { "provider": "groq", "source": { "url": "https://console.groq.com/docs/models.md", "fetchedAt": "2026-08-19T00:56:02.388Z" }, "models": [ { "id": "GPT OSS 120Bopenai/gpt-oss-120b", "name": "GPT OSS 120Bopenai/gpt-oss-120b", "provider": "groq", "tiers": [ { "name": "standard", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.15 input" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.60 output" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "contextWindow": 131072 } }, { "id": "GPT OSS 20Bopenai/gpt-oss-20b", "name": "GPT OSS 20Bopenai/gpt-oss-20b", "provider": "groq", "tiers": [ { "name": "standard", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.075 input" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.30 output" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "contextWindow": 131072 } }, { "id": "Whisperwhisper-large-v3", "name": "Whisperwhisper-large-v3", "provider": "groq", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.111, "currency": "USD", "units": 1, "pricingType": "hour", "raw": "$0.111 per hour" } ] } ], "metadata": { "contextWindow": null } }, { "id": "Whisper Large V3 Turbowhisper-large-v3-turbo", "name": "Whisper Large V3 Turbowhisper-large-v3-turbo", "provider": "groq", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "hour", "raw": "$0.04 per hour" } ] } ], "metadata": { "contextWindow": null } }, { "id": "Compoundgroq/compound", "name": "Compoundgroq/compound", "provider": "groq", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "contextWindow": 131072 } }, { "id": "Compound Minigroq/compound-mini", "name": "Compound Minigroq/compound-mini", "provider": "groq", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "contextWindow": 131072 } }, { "id": "Canopy Labs Orpheus Arabic Saudicanopylabs/orpheus-arabic-saudi", "name": "Canopy Labs Orpheus Arabic Saudicanopylabs/orpheus-arabic-saudi", "provider": "groq", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "character", "raw": "$40.00 per 1M characters" } ] } ], "metadata": { "contextWindow": 4000 } }, { "id": "Canopy Labs Orpheus V1 Englishcanopylabs/orpheus-v1-english", "name": "Canopy Labs Orpheus V1 Englishcanopylabs/orpheus-v1-english", "provider": "groq", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "character", "raw": "$22.00 per 1M characters" } ] } ], "metadata": { "contextWindow": 4000 } }, { "id": "Llama Prompt Guard 2 22Mmeta-llama/llama-prompt-guard-2-22m", "name": "Llama Prompt Guard 2 22Mmeta-llama/llama-prompt-guard-2-22m", "provider": "groq", "tiers": [ { "name": "standard", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.03 input" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.03 output" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "contextWindow": 512 } }, { "id": "Prompt Guard 2 86Mmeta-llama/llama-prompt-guard-2-86m", "name": "Prompt Guard 2 86Mmeta-llama/llama-prompt-guard-2-86m", "provider": "groq", "tiers": [ { "name": "standard", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.04 input" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.04 output" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "contextWindow": 512 } }, { "id": "MiniMax M2.7Enterpriseminimaxai/minimax-m2.7", "name": "MiniMax M2.7Enterpriseminimaxai/minimax-m2.7", "provider": "groq", "tiers": [ { "name": "standard", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "contextWindow": 196608 } }, { "id": "Safety GPT OSS 20Bopenai/gpt-oss-safeguard-20b", "name": "Safety GPT OSS 20Bopenai/gpt-oss-safeguard-20b", "provider": "groq", "tiers": [ { "name": "standard", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.075 input" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.30 output" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "contextWindow": 131072 } }, { "id": "Qwen/Qwen3.6-27Bqwen/qwen3.6-27b", "name": "Qwen/Qwen3.6-27Bqwen/qwen3.6-27b", "provider": "groq", "tiers": [ { "name": "standard", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$0.60 input" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "$3.00 output" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "contextWindow": 131072 } } ] }, { "provider": "doubleword", "source": { "url": "https://docs.doubleword.ai/api/models", "fetchedAt": "2026-08-19T00:56:02.725Z" }, "models": [ { "id": "Qwen/Qwen3.8-27B-FP8", "name": "Qwen3.8 27B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "Alibaba", "type": "Generation", "capabilities": [ "reasoning", "vision" ], "description": "Qwen3.8-27B is a compact multimodal reasoning model from Alibaba’s Qwen family, designed for general-purpose reasoning, coding, tool use, and vision workloads. Its 262K-token context window supports long documents and extended agentic tasks, while the FP8 deployment offers efficient serving.\n\n---\n**Thinking Mode:**\n\nThis model reasons step-by-step before responding by default.\n\nSet `reasoning_effort` to `none` to disable thinking. Other effort levels enable thinking but do not select graduated reasoning budgets." } }, { "id": "meta-models/Muse-Glimmer-30B", "name": "Muse Glimmer 30B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "meta", "type": "Generation", "capabilities": [ "vision", "reasoning" ], "description": "Muse Glimmer is a 30-billion-parameter language model with a dedicated image encoder, distilled from Muse Spark and purpose-built for autonomous agentic tasks on consumer hardware. The model integrates multi-step reasoning, reliable tool use, multimodal understanding, and failure recovery into a small model.\n\n----\n**Reasoning:** This model controls reasoning via prompting, not request parameters. Reasoning strength can be defined as part of the system prompt as Reasoning strength: . Muse Glimmer supports the following levels: low / medium / high / xhigh. Use high or xhigh for complex problem solving, coding, and agentic tasks.", "cachePricing": { "enabled": true, "readMultiplier": 0.12, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-08-12T12:14:42.493740Z", "validUntil": null } } }, { "id": "deepseek-ai/DeepSeek-V4-Flash-0731", "name": "DeepSeek V4 Flash 0731", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "deepseek", "type": "Generation", "capabilities": [ "reasoning" ], "description": "DeepSeek V4-Flash 0731 is the latest release in DeepSeek’s V4 family and delivers a major upgrade for agent use cases. With 284B total parameters, 13B activated per token, and a 1M-token context window, it combines efficient inference with substantially stronger results than V4-Flash Preview and V4-Pro Preview.", "cachePricing": { "enabled": true, "readMultiplier": 0.2, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-08-12T12:15:22.107073Z", "validUntil": null } } }, { "id": "moonshotai/kimi-k3", "name": "Kimi K3", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 2.1500000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 11.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "moonshot", "type": "Generation", "capabilities": [ "reasoning", "vision" ], "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at navigating large repositories, using tools, debugging, and iterating against images, logs, tests, and runtime feedback. Its architecture uses KDA and Attention Residuals for computational efficiency.", "cachePricing": { "enabled": true, "readMultiplier": 0.1, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-08-12T12:15:52.533300Z", "validUntil": null } } }, { "id": "thinkingmachines/Inkling-NVFP4", "name": "Inkling", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "thinkingmachines", "type": "Generation", "capabilities": [ "vision", "reasoning" ], "description": "Inkling is the first model released by Thinking Machines - a large Mixture of Experts model with 975B parameters, 41B active, trained on over 45T tokens of text, image, and audio data. It supports up to a 1M context length and and offers strong performance across the board for agentic and reasoning heavy tasks.", "cachePricing": { "enabled": true, "readMultiplier": 0.17, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "tencent/Hy3-FP8", "name": "Hy3", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.58, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "tencent", "type": "Generation", "capabilities": [ "reasoning" ], "description": "Hy3 is a 295B-parameter Mixture-of-Experts (MoE) model with 21B active parameters, developed by the Tencent Hy Team. Hy3 outperforms similar-size models and rivals flagship open-source models with 2-5x fewer parameters." } }, { "id": "zai-org/GLM-5.2-FP8", "name": "GLM 5.2", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.47, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.9299999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "zai", "type": "Generation", "capabilities": [ "reasoning" ], "description": " Meet GLM-5.2-FP8 - Z.ai’s latest flagship open model for long-horizon agentic work, coding, and complex engineering tasks. GLM-5.2 delivers a major step up from GLM-\n 5.1, pairing stronger real-world coding performance with a solid 1M-token context window for sustained repository-scale workflows. It performs especially well on\n agentic engineering benchmarks, including SWE-bench Pro, NL2Repo, DeepSWE, Terminal Bench 2.1, FrontierSWE, and SWE-Marathon, making it well suited for extended coding\n sessions, terminal use, repository generation, debugging, tool orchestration, and ambiguous multi-step projects.\n\n GLM-5.2-FP8 uses Z.ai’s improved GLM MoE architecture with FP8 quantization, IndexShare sparse-attention optimization, and enhanced speculative decoding via improved\n MTP, reducing long-context compute while improving throughput.\n\n Best for:\n\n Agentic engineering and complex software development\n Long-horizon coding, debugging, and terminal workflows\n Large-context repository understanding and generation\n Tool-use agents that need sustained iteration over long sessions\n Ambiguous engineering tasks requiring planning, experiments, and judgment\n Open-weight deployment where FP8 efficiency matters\n\n Max Total Tokens: 1048576\n\n Sampling Parameters:\n\n We have set the default sampling parameters using the recommended values from the GLM-5.2-FP8 generation configuration:\n\n Temperature=1.0 and TopP=0.95.\n\n You can adjust these on a per-request basis by setting the sampling parameters in the request body.\n\n Thinking Mode:\n\n This model is designed for long-horizon reasoning and supports flexible thinking effort levels to balance latency and quality. It is especially useful when the task\n requires planning, reading code, running commands, interpreting results, identifying blockers, and iterating toward a working solution.\n\n", "cachePricing": { "enabled": true, "readMultiplier": 0.2, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4", "name": "Nemotron 3 Ultra 550B A55B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.6500000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "nvidia", "type": "Generation", "capabilities": [ "reasoning" ], "description": "NVIDIA Nemotron 3 Ultra is NVIDIA’s largest open Nemotron 3 model, built for advanced reasoning, agentic workflows, tool use, and knowledge-intensive tasks. With 550B total parameters, 55B active parameters, and NVFP4 weights, it delivers frontier-scale capability in a sparse model design. It is well suited to complex problem solving, coding, research assistance, multilingual chat, and high-stakes RAG.", "cachePricing": { "enabled": true, "readMultiplier": 0.25, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.98, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 1.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 2.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "deepseek", "type": "Generation", "capabilities": [ "reasoning" ], "description": "DeepSeek V4-Pro is DeepSeek’s flagship open MoE model for advanced reasoning, coding, and agentic work. With 1.6T total parameters, 49B active parameters, and a 1M-token context window, it is designed for the hardest tasks in the V4 lineup: complex problem solving, knowledge-intensive workflows, and high-stakes coding or research use cases.\n", "cachePricing": { "enabled": true, "readMultiplier": 0.1, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-01T21:30:23.138860Z", "validUntil": null } } }, { "id": "deepseek-ai/DeepSeek-V4-Flash", "name": "DeepSeek V4 Flash", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "deepseek", "type": "Generation", "capabilities": [ "reasoning" ], "description": "DeepSeek V4-Flash is a general-purpose open MoE model built for reasoning, tool use, and long-context work. With 284B total parameters, 13B active parameters, and a 1M-token context window, it brings the core strengths of the V4 family into a more compact package. It’s a strong fit for chat, structured generation, document-scale analysis, and agentic workflows that need broad capability across everyday tasks.", "cachePricing": { "enabled": true, "readMultiplier": 0.2, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "Qwen/Qwen3.6-35B-A3B-FP8", "name": "Qwen3.6 35B A3B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "alibaba", "type": "Generation", "capabilities": [ "reasoning", "vision" ], "description": "Qwen3.6-35B-A3B is an updated version of the Qwen3.5-35B-A3B model, prioritizing stability and real-world utility following community feedback. It is a high-intelligence, mid-sized model that hits a very compelling price/performance point for async workloads. In Qwen's published benchmarks, this model outperformed GPT-5-mini, GPT-OSS-120B, and Claude Sonnet 4.5.\n\n---\n**Thinking Mode:**\n\nThis model reasons step-by-step before responding by default. This model does not support graduated thinking levels. Parameters such as reasoning_effort are not supported and will have no effect.\n\n---", "cachePricing": { "enabled": true, "readMultiplier": 1, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "moonshotai/Kimi-K2.6", "name": "Kimi K2.6", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 2.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.71, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 3.41, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "moonshot", "type": "Generation", "capabilities": [ "reasoning", "vision" ], "description": "Kimi K2.6 is an open-source, native multimodal agentic model that advances practical capabilities in long-horizon coding, coding-driven design, proactive autonomous execution, and swarm-based task orchestration. Built on the same MoE multimodal architecture as K2.5 with a 256K context window, K2.6 combines strong reasoning, visual understanding, and agentic tool use across instant and thinking modes.\n\nKey Features\nLong-Horizon Coding: K2.6 improves end-to-end coding performance across Rust, Go, Python, front-end, DevOps, and performance optimization workflows.\nCoding-Driven Design: K2.6 turns prompts and visual inputs into production-ready interfaces and lightweight full-stack workflows with structured layouts, interactive elements, and deliberate visual polish.\nElevated Agent Swarm: K2.6 can decompose complex tasks into parallel, domain-specialized subtasks, scaling to large coordinated agent runs for end-to-end outputs.\nProactive Orchestration: K2.6 is built for autonomous execution, supporting persistent background agents that manage schedules, execute code, and coordinate cross-platform operations with minimal oversight.", "cachePricing": { "enabled": true, "readMultiplier": 0.2, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "zai-org/GLM-5.1-FP8", "name": "GLM 5.1", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.7899999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 2.63, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.53, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 1.0499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "zai", "type": "Generation", "capabilities": [ "reasoning" ], "description": "Meet **GLM-5.1-FP8** - Z.ai's next-generation flagship model for agentic engineering, with significantly stronger coding capabilities than GLM-5. It achieves state-of-the-art performance on SWE-Bench Pro and leads GLM-5 by a wide margin on NL2Repo and Terminal-Bench 2.0, making it especially strong for real-world coding, repository generation, terminal tasks, and long-horizon agentic workflows. GLM-5.1 is designed to stay productive over extended sessions, breaking down ambiguous problems, running experiments, reading results, identifying blockers, and improving through repeated iteration.\n\nBest for:\n* Agentic engineering and complex coding tasks\n* Long-running tool-use workflows\n* Repository generation and terminal-based development\n* Ambiguous problems requiring judgment, experimentation, and sustained reasoning\n\n\n**Max Total Tokens:** 202752\n\n**Sampling Parameters:**\n\nWe have set the default sampling parameters using the recommended values from the GLM-5.1 generation configuration:\n\n---\n\n_Temperature=1.0 and TopP=0.95._\n\n---\n\nYou can adjust these on a per-request basis by setting the sampling parameters in the request body.\n\n---\n\n**Thinking Mode:**\n\nThis model reasons step-by-step before responding by default.\n\nThis model is built for long-horizon agentic work and can sustain useful reasoning over extended sessions involving planning, tool use, experiments, and iterative debugging.", "cachePricing": { "enabled": true, "readMultiplier": 0.25, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "google/gemma-4-31B-it", "name": "Gemma 4 31B IT", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "google", "type": "Generation", "capabilities": [ "vision", "reasoning" ], "description": "**Gemma 4 31B** is Google DeepMind’s most capable open model, built for advanced reasoning, coding, and multimodal understanding. It sits in the same general tier as **Claude 4.5 Haiku**\n and **NVIDIA Nemotron 3 Super**, with native function calling and structured JSON output for agentic workflows; strong image and video understanding for tasks like OCR and chart analysis;\n **256K context** for long documents and repositories; and support for **140+ languages**.\n\n———\n\n**Multimodal Input**\n\nGemma 4 supports multimodal input, so you can send images or videos together with text in a single request.\n\n*Image Example*\n\n```\n\"messages\": [\n {\n \"role\": \"user\",\n \"content\": [\n {\n \"type\": \"image_url\",\n \"image_url\": {\n \"url\": \"https://example.com/image.jpg\"\n }\n },\n {\n \"type\": \"text\",\n \"text\": \"Describe this image.\"\n }\n ]\n }\n]\n```\n\n*Video Example*\n\n```\n\"messages\": [\n {\n \"role\": \"user\",\n \"content\": [\n {\n \"type\": \"video_url\",\n \"image_url\": {\n \"url\": \"https://example.com/sample_video.mp4\"\n }\n },\n {\n \"type\": \"text\",\n \"text\": \"Summarize what happens in this video.\"\n }\n ]\n }\n]\n```", "cachePricing": { "enabled": true, "readMultiplier": 1, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "google/gemma-4-26B-A4B-it", "name": "google/gemma-4-26B-A4B-it", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "google", "type": "Generation", "capabilities": [ "vision", "reasoning" ], "description": "Gemma 4 26B-A4B is one of Google DeepMind’s most capable open models, built for advanced reasoning, coding, and multimodal understanding. It uses an MoE arcitechture for more efficient inference. It sits in the same general tier as Claude 4.5 Haiku and NVIDIA Nemotron 3 Super, with native function calling and structured JSON output for agentic workflows; strong image and video understanding for tasks like OCR and chart analysis; 256K context for long documents and repositories; and support for 140+ languages.\n\n", "cachePricing": { "enabled": true, "readMultiplier": 0.25, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", "name": "Nemotron 3 Super 120B A12B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.33999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.22999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "NVIDIA", "type": "Generation", "capabilities": [ "reasoning" ], "description": "NVIDIA Nemotron 3 Super 120B A12B NVFP4 is an open hybrid Mamba-Transformer LatentMoE model with 120 billion total parameters and 12 billion active parameters, built for agentic reasoning workloads such as coding, planning, tool use, and long-context tasks. It sits in the same capability tier as Qwen3.5-122B non-reasoning and ahead of GPT-OSS-120B, while also delivering higher throughput.\n\n---\n\nIn line with NVIDIA's guidance, we use `temperature=1.0` and `top_p=0.95` across all tasks and serving backends, including reasoning, tool calling, and general chat.", "cachePricing": { "enabled": true, "readMultiplier": 0.3, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "Qwen/Qwen3.5-9B-dottxt", "name": "Qwen3.5 9B dottxt", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.58, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "alibaba", "type": "Generation", "capabilities": [ "reasoning", "enhanced_structured_generation", "vision" ], "description": "Qwen3.5-9B is a compact 9B parameter reasoning model with a 262K token native context length, designed for strong reasoning performance while remaining extremely cost-efficient. Despite its small size, it performs remarkably well on complex tasks and in Qwen's benchmarks outperformed the, much larger, GPT-OSS-120 model.\n\n---\n**Thinking Mode:** \n\nThis model reasons step-by-step before responding by default.\n\nThis model does not support graduated thinking levels. Parameters such as `reasoning_effort` are not supported and will have no effect.", "cachePricing": { "enabled": true, "readMultiplier": 0.25, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "Qwen/Qwen3.5-4B", "name": "Qwen3.5 4B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "Alibaba", "type": "Generation", "capabilities": [ "vision", "reasoning" ], "description": "Qwen3.5-4B is a compact open 4B model with a native 262K context window, designed to deliver strong reasoning, coding, and long-context performance in a very small footprint. Qwen reports that it outperforms GPT-OSS-20B across several key benchmarks, including MMLU-Pro, GPQA Diamond, AA-LCR, and LongBench v2, making it a standout small model for cost-sensitive workloads.\n\n---\n**Thinking Mode:** \n\nThis model reasons step-by-step before responding by default.\n\nThis model does not support graduated thinking levels. Parameters such as `reasoning_effort` are not supported and will have no effect.", "cachePricing": { "enabled": true, "readMultiplier": 0.25, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "Qwen/Qwen3.5-9B", "name": "Qwen3.5 9B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "Alibaba", "type": "Generation", "capabilities": [ "vision", "reasoning" ], "description": "Qwen3.5-9B is a compact 9B parameter reasoning model with a 262K token native context length, designed for strong reasoning performance while remaining extremely cost-efficient. Despite its small size, it performs remarkably well on complex tasks and in Qwen's benchmarks outperformed the, much larger, GPT-OSS-120 model.\n\n---\n**Thinking Mode:** \n\nThis model reasons step-by-step before responding by default.\n\nThis model does not support graduated thinking levels. Parameters such as `reasoning_effort` are not supported and will have no effect.", "cachePricing": { "enabled": true, "readMultiplier": 0.25, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "Qwen/Qwen3.5-35B-A3B-FP8-dottxt", "name": "Qwen3.5 35B A3B dottxt", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "alibaba", "type": "Generation", "capabilities": [ "reasoning", "enhanced_structured_generation", "vision" ], "description": "Qwen3.5-35B-A3B is a high-intelligence, mid-sized model that hits a very compelling price/performance point for async workloads. In Qwen's published benchmarks, this model outperformed GPT-5-mini, GPT-OSS-120B, and Claude Sonnet 4.5. \n\n----- \n**Thinking Mode:** \n\nThis model reasons step-by-step before responding by default.\n\nThis model does not support graduated thinking levels. Parameters such as `reasoning_effort` are not supported and will have no effect.", "cachePricing": { "enabled": true, "readMultiplier": 0.5714, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-08-14T15:06:40.707047Z", "validUntil": null } } }, { "id": "Qwen/Qwen3.5-397B-A17B-FP8-dottxt", "name": "Qwen3.5 397B A17B dottxt", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "alibaba", "type": "Generation", "capabilities": [ "vision", "reasoning", "enhanced_structured_generation" ], "description": "Meet **Qwen3.5-397B-A17B** - released Feb 2026, it is Qwen's most powerful model, delivering performance similar to GPT-5.2 and Claude Opus 4.5 on challenging tasks including advanced reasoning, mathematics, and complex code generation. Offers frontier-level capabilities at a fraction of the cost.\nBest for:\n* Tasks requiring maximum intelligence\n* Complex analysis\n* Sophisticated coding projects\n* Scenarios where quality justifies the additional cost over smaller models\n\n\n**Max New Tokens:** 16384\n\n**Max Total Tokens:** 262144\n\n**Sampling Parameters:**\n\nWe have set the default sampling parameters using the recommended values set out by the Qwen team:\n\n---\n\n_We suggest using Temperature=0.7, TopP=0.8, TopK=20, and MinP=0._\n \n_For supported frameworks, you can adjust the presence_penalty parameter between 0 and 2 to reduce endless repetitions. However, using a higher value may occasionally result in language mixing and a slight decrease in model performance._\n\n---\n\n\n \nWe use a **default presence_penalty** of 1.5 to bias the model against endless repetitions, if you still notice this behaviour try increasing the presence_penalty.\n\nYou can adjust these on a per-request basis by setting the sampling parameters in the request body. \n", "cachePricing": { "enabled": true, "readMultiplier": 0.4, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "Qwen/Qwen3.5-397B-A17B-FP8", "name": "Qwen3.5 397B A17B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.8399999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.23, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.39, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 2.4499999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "Alibaba", "type": "Generation", "capabilities": [ "vision", "reasoning" ], "description": "Meet **Qwen3.5-397B-A17B** - released Feb 2026, it is Qwen's most powerful model, delivering performance similar to GPT-5.2 and Claude Opus 4.5 on challenging tasks including advanced reasoning, mathematics, and complex code generation. Offers frontier-level capabilities at a fraction of the cost.\nBest for:\n* Tasks requiring maximum intelligence\n* Complex analysis\n* Sophisticated coding projects\n* Scenarios where quality justifies the additional cost over smaller models\n\n\n**Max New Tokens:** 16384\n\n**Max Total Tokens:** 262144\n\n**Sampling Parameters:**\n\nWe have set the default sampling parameters using the recommended values set out by the Qwen team:\n\n---\n\n_We suggest using Temperature=0.7, TopP=0.8, TopK=20, and MinP=0._\n \n_For supported frameworks, you can adjust the presence_penalty parameter between 0 and 2 to reduce endless repetitions. However, using a higher value may occasionally result in language mixing and a slight decrease in model performance._\n\n---\n\n\n \nWe use a **default presence_penalty** of 1.5 to bias the model against endless repetitions, if you still notice this behaviour try increasing the presence_penalty.\n\nYou can adjust these on a per-request basis by setting the sampling parameters in the request body. \n", "cachePricing": { "enabled": true, "readMultiplier": 0.4, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "deepseek-ai/DeepSeek-OCR-2", "name": "DeepSeek OCR 2", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "DeepSeek", "type": "OCR", "capabilities": [ "vision" ], "description": "Meet DeepSeek-OCR-2 - Deepseek's latest OCR model. This model expands on Deepseek-OCR with a novel causal vision encoder that captures reading order to enhance structured extraction of text.\n\n---\n\n**Usage Tips**\n\nUse `Free OCR.` for plain text extraction when you want only the text content from the image, without preserving layout or structure.\n```\nmessages = [{\"role\": \"user\", \"content\": [{\"type\": \"text\", \"text\": \"Free OCR.\"}, {\"type\": \"image_url\", \"image_url\": {\"url\": image_url}}]}]\n```\n\nUse `<|grounding|>Convert the document to markdown.` for structured markdown extraction when you want to preserve headings, paragraphs, lists, and tables.\n```\nmessages = [{\"role\": \"user\", \"content\": [{\"type\": \"text\", \"text\": \"<|grounding|>Convert the document to markdown.\"}, {\"type\": \"image_url\", \"image_url\": {\"url\": image_url}}]}]\n```\n" } }, { "id": "lightonai/LightOnOCR-2-1B-bbox-soup", "name": "LightOnOCR 2 1B bbox soup", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "LightOn", "type": "OCR", "capabilities": [ "vision" ], "description": "LightOnOCR-2 is an efficient end-to-end 1B-parameter vision-language model for converting documents (PDFs, scans, images) into clean, naturally ordered text without relying on brittle pipelines. This second version is trained on a larger and higher-quality corpus with stronger French, arXiv, and scan coverage, improved LaTeX handling, and cleaner normalization. LightOnOCR-2 achieves state-of-the-art performance on OlmOCR-Bench while being ~9× smaller and significantly faster than competing approaches.\n\n*Merged bbox variant:* This model combines OCR-improving RLVR signals with bounding-box-focused RLVR updates via joint merging, preserving OCR quality while providing image localization.\n\n---\n**Usage Tips**\n\nDo not include a system prompt or user prompt as the model has a tendency to repeat the prompt in its answer. \n\n```\npayload = {\n \"model\": \"lightonai/LightOnOCR-2-1B-bbox-soup\",\n \"messages\": [{\n \"role\": \"user\",\n \"content\": [{\n \"type\": \"image_url\",\n \"image_url\": {\"url\": f\"data:image/png;base64,{image_base64}\"}\n }]\n }],\n \"max_tokens\": 4096,\n \"temperature\": 0.2,\n \"top_p\": 0.9,\n}\n```\n\n#### Rendering and Preprocessing Tips\n- Render PDFs to PNG or JPEG at a target longest dimension of 1540px\n- Maintain aspect ratio to preserve text geometry\n- Use one image per page\n" } }, { "id": "allenai/olmOCR-2-7B-1025-FP8", "name": "olmOCR 2 7B 1025", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "Ai2", "type": "OCR", "capabilities": [ "vision" ], "description": "This is a release of the olmOCR model that's fine tuned from Qwen2.5-VL-7B-Instruct using the olmOCR-mix-1025 dataset. It has been additionally fine tuned using GRPO RL training to boost its performance at math equations, tables, and other tricky OCR cases.\n\n---\n**Usage Tips**\n\nThis model expects a prompt alongside the image. The default one used in the olmocr repo is this:\n\n```\nprompt = \"Attached is one page of a document that you must process. Just return the plain text representation of this document as if you were reading it naturally. Convert equations to LateX and tables to HTML.\\nIf there are any figures or charts, label them with the following markdown syntax ![Alt text describing the contents of the figure](page_startx_starty_width_height.png)\\nReturn your output as markdown, with a front matter section on top specifying values for the primary_language, is_rotation_valid, rotation_correction, is_table, and is_diagram parameters\"\n\nmessages = [\n {\n \"role\": \"user\",\n \"content\": [\n {\"type\": \"text\", \"text\": prompt},\n {\"type\": \"image_url\", \"image_url\": {\"url\": f\"data:image/png;base64,{image_base64}\"}},\n ],\n }\n ]\n\n```\n\n\n**Image Processing**\n\nThis model expects as input a single document image, rendered such that the longest dimension is 1288 pixels." } }, { "id": "Qwen/Qwen3-VL-30B-A3B-Instruct-FP8", "name": "Qwen3 VL 30B A3B Instruct", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "Alibaba", "type": "Generation", "capabilities": [ "vision" ], "description": "Meet **Qwen3-VL-30B**, the smaller model of the Qwen3-VL family, delivering performance similar to GPT-4.1-mini and Claude Sonnet 4. This highly capable mid-size model is suited for tasks that are constrained or require high token volumes. Excels at reasoning, coding, and structured output generation. \n\nBest for:\n- Production workloads requiring strong performance without frontier model costs\n- Complex reasoning tasks\n- Code generation", "cachePricing": { "enabled": true, "readMultiplier": 0.25, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "Qwen/Qwen3-VL-235B-A22B-Instruct-FP8", "name": "Qwen3 VL 235B A22B Instruct", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.4300000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 1.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "Alibaba", "type": "Generation", "capabilities": [ "vision" ], "description": "Meet **Qwen3-VL-235B** - delivering performance similar to GPT-5 Chat and Claude 4 Opus Thinking on challenging tasks including advanced reasoning, mathematics, and complex code generation. Offers frontier-level capabilities at a fraction of the cost.\nBest for:\n* Tasks requiring maximum intelligence\n* Complex analysis\n* Sophisticated coding projects\n* Scenarios where quality justifies the additional cost over smaller models\n\n\n**Max New Tokens:** 16384\n\n**Max Total Tokens:** 262144\n\n**Sampling Parameters:**\n\nWe have set the default sampling parameters using the recommended values set out by the Qwen team:\n\n---\n\n_We suggest using Temperature=0.7, TopP=0.8, TopK=20, and MinP=0._\n \n_For supported frameworks, you can adjust the presence_penalty parameter between 0 and 2 to reduce endless repetitions. However, using a higher value may occasionally result in language mixing and a slight decrease in model performance._\n\n---\n\n\n \nWe use a **default presence_penalty** of 1.5 to bias the model against endless repetitions, if you still notice this behaviour try increasing the presence_penalty.\n\nYou can adjust these on a per-request basis by setting the sampling parameters in the request body. \n", "cachePricing": { "enabled": true, "readMultiplier": 0.55, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "openai", "type": "Generation", "capabilities": [ "reasoning" ], "description": "Meet **gpt-oss-20b** — OpenAI’s larger open-weight MoE model, with 117B total parameters, 5.1B active parameters, and a 128k-token context window. ", "cachePricing": { "enabled": true, "readMultiplier": 0.75, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 512, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "openai/gpt-oss-20b", "name": "GPT OSS 20B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "OpenAI", "type": "Generation", "capabilities": [ "reasoning" ], "description": "Meet **gpt-oss-20b** — for lower latency, and local or specialized use cases (21B parameters with 3.6B active parameters)", "cachePricing": { "enabled": true, "readMultiplier": 0.65, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } }, { "id": "Qwen/Qwen3-Embedding-8B", "name": "Qwen3 Embedding 8B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "Alibaba", "type": "Embedding", "capabilities": [], "description": "The Qwen3 Embedding model series is the latest embeddings model in the Qwen family, specifically designed for text embedding and ranking tasks. Building upon the dense foundational models of the Qwen3 series, it provides a comprehensive range of text embeddings and reranking models in various sizes (0.6B, 4B, and 8B). This series inherits the exceptional multilingual capabilities, long-text understanding, and reasoning skills of its foundational model. The Qwen3 Embedding series represents significant advancements in multiple text embedding and ranking tasks, including text retrieval, code retrieval, text classification, text clustering, and bitext mining.\n\n**Exceptional Versatility**: The embedding model has achieved state-of-the-art performance across a wide range of downstream application evaluations. The 8B size embedding model ranks **No.1** in the MTEB multilingual leaderboard (as of June 5, 2025, score **70.58**), while the reranking model excels in various text retrieval scenarios.\n\n**Comprehensive Flexibility**: The Qwen3 Embedding series offers a full spectrum of sizes (from 0.6B to 8B) for both embedding and reranking models, catering to diverse use cases that prioritize efficiency and effectiveness. Developers can seamlessly combine these two modules. Additionally, the embedding model allows for flexible vector definitions across all dimensions, and both embedding and reranking models support user-defined instructions to enhance performance for specific tasks, languages, or scenarios.\n\n**Multilingual Capability**: The Qwen3 Embedding series offer support for over 100 languages, thanks to the multilingual capabilites of Qwen3 models. This includes various programming languages, and provides robust multilingual, cross-lingual, and code retrieval capabilities.\n\n**Qwen3-Embedding-8B** has the following features:\n\n- Model Type: Text Embedding\n- Supported Languages: 100+ Languages\n- Number of Paramaters: 8B\n- Context Length: 32k\n- Embedding Dimension: Up to 4096, supports user-defined output dimensions ranging from 32 to 4096\n\nFor more details, including benchmark evaluation, hardware requirements, and inference performance, please refer to our [blog](https://qwenlm.github.io/blog/qwen3-embedding/), [GitHub](https://github.com/QwenLM/Qwen3-Embedding)." } }, { "id": "Qwen/Qwen3-14B-FP8", "name": "Qwen3 14B", "provider": "doubleword", "tiers": [ { "name": "async", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "batch24h", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] }, { "name": "realtime", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "providerName": "Alibaba", "type": "Generation", "capabilities": [], "description": "Meet **Qwen3-14B** - a small text-only model from the Qwen3 release. \n\nBest for:\n* High volume tasks\n* Tasks that do not require maximum performance, such as classification, extraction, or summarization\n\n\n**Max New Tokens:** 16384\n\n**Max Total Tokens:** 262144\n\n**Sampling Parameters:**\n\nWe have set the default sampling parameters using the recommended values set out by the Qwen team:\n\n---\n\n_We suggest using Temperature=0.7, TopP=0.8, TopK=20, and MinP=0._\n \n_For supported frameworks, you can adjust the presence_penalty parameter between 0 and 2 to reduce endless repetitions. However, using a higher value may occasionally result in language mixing and a slight decrease in model performance._\n\n---\n\n\n \nWe use a **default presence_penalty** of 1.5 to bias the model against endless repetitions, if you still notice this behaviour try increasing the presence_penalty.\n\nYou can adjust these on a per-request basis by setting the sampling parameters in the request body. \n", "cachePricing": { "enabled": true, "readMultiplier": 0.25, "writeMultiplier5m": 1.25, "writeMultiplier1h": 2, "writeMultiplier24h": 2.5, "minPrefixTokens": 1024, "validFrom": "2026-07-31T12:34:51.847292Z", "validUntil": null } } } ] }, { "provider": "openrouter", "source": { "url": "https://openrouter.ai/api/v1/models?output_modalities=all", "fetchedAt": "2026-08-19T00:56:02.725Z" }, "models": [ { "id": "z-ai/glm-5.3@z.ai", "name": "Z.ai: GLM 5.3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.3", "canonicalSlug": "z-ai/glm-5.3-20260816", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-5.3-20260816", "model_id": "z-ai/glm-5.3", "model_name": "Z.ai: GLM 5.3", "context_length": 1048576, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-27b@akashml", "name": "Qwen: Qwen3.8 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-27b", "canonicalSlug": "qwen/qwen3.8-27b-20260814", "servingProvider": "AkashML", "servingProviderSlug": "akashml", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "max_tokens", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...", "uptimeLast30m": 99.70315398886828, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AkashML | qwen/qwen3.8-27b-20260814", "model_id": "qwen/qwen3.8-27b", "model_name": "Qwen: Qwen3.8 27B", "context_length": 262144, "pricing": { "prompt": "0.00000045", "completion": "0.0000032", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "AkashML", "tag": "akashml/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "max_tokens", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.70315398886828, "uptime_last_5m": 99.24050632911391, "uptime_last_1d": 96.53980723578445, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-27b@venice", "name": "Qwen: Qwen3.8 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-27b", "canonicalSlug": "qwen/qwen3.8-27b-20260814", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...", "uptimeLast30m": 99.85218033998522, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | qwen/qwen3.8-27b-20260814", "model_id": "qwen/qwen3.8-27b", "model_name": "Qwen: Qwen3.8 27B", "context_length": 262144, "pricing": { "prompt": "0.00000045", "completion": "0.0000032", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.85218033998522, "uptime_last_5m": 99.02912621359224, "uptime_last_1d": 99.92884428132724, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-27b@chutes", "name": "Qwen: Qwen3.8 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [ { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000045" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-27b", "canonicalSlug": "qwen/qwen3.8-27b-20260814", "servingProvider": "Chutes", "servingProviderSlug": "chutes", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...", "uptimeLast30m": 99.08735332464146, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Chutes | qwen/qwen3.8-27b-20260814", "model_id": "qwen/qwen3.8-27b", "model_name": "Qwen: Qwen3.8 27B", "context_length": 262144, "pricing": { "prompt": "0.00000045", "completion": "0.0000032", "input_cache_read": "0.000000045", "discount": 0 }, "provider_name": "Chutes", "tag": "chutes/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.08735332464146, "uptime_last_5m": 100, "uptime_last_1d": 99.16878404242372, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-27b@io-net", "name": "Qwen: Qwen3.8 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000048" } ], "output": [ { "amount": 3.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000034" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-27b", "canonicalSlug": "qwen/qwen3.8-27b-20260814", "servingProvider": "Io Net", "servingProviderSlug": "io-net", "contextLength": 65500, "maxCompletionTokens": 65536, "quantization": "fp8", "status": -5, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "seed", "structured_outputs", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...", "uptimeLast30m": 71.60493827160494, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Io Net | qwen/qwen3.8-27b-20260814", "model_id": "qwen/qwen3.8-27b", "model_name": "Qwen: Qwen3.8 27B", "context_length": 65500, "pricing": { "prompt": "0.00000048", "completion": "0.0000034", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "Io Net", "tag": "io-net/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "seed", "structured_outputs", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "status": -5, "uptime_last_30m": 71.60493827160494, "uptime_last_5m": 100, "uptime_last_1d": 86.58841158841159, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "dots-studio/dots-3-note-preview:free@atlascloud", "name": "Dots Studio: Dots3-Note Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "dots-studio/dots-3-note-preview:free", "canonicalSlug": "dots-studio/dots-3-note-preview-20260813", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 512000, "maxCompletionTokens": 512000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Dots3-Note Preview is an open-weight mixture-of-experts model from Dots Studio, with 16B active parameters out of 280B total. It is the lightest model in the Dots 3 family and is...", "uptimeLast30m": 99.52849528495284, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | dots-studio/dots-3-note-preview-20260813:free", "model_id": "dots-studio/dots-3-note-preview:free", "model_name": "Dots Studio: Dots3-Note Preview", "context_length": 512000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 512000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.52849528495284, "uptime_last_5m": 99.58720330237358, "uptime_last_1d": 99.58312982319039, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b@deepinfra", "name": "NVIDIA: Nemotron 3.5 ASR Streaming Multilingual 0.6B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000333" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b", "canonicalSlug": "nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b-20260813", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Other", "instruct_type": null }, "description": "Nemotron 3.5 ASR Streaming Multilingual 0.6B is a speech recognition model from NVIDIA. Its prompt-conditioned, cache-aware FastConformer-RNNT design targets low-latency transcription across more than 40 languages for real-time captioning, voice...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b-20260813", "model_id": "nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b", "model_name": "NVIDIA: Nemotron 3.5 ASR Streaming Multilingual 0.6B", "context_length": 0, "pricing": { "prompt": "0.00000333", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/voxtral-small-24b-2507-stt@deepinfra", "name": "Mistral: Voxtral Small 24B 2507 STT", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/voxtral-small-24b-2507-stt", "canonicalSlug": "mistralai/voxtral-small-24b-2507-stt-20260813", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 0, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Voxtral Small 24B 2507 STT is a speech transcription model from Mistral AI. It is suited for transcription, translation, and audio understanding workloads that benefit from its larger model capacity.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | mistralai/voxtral-small-24b-2507-stt-20260813", "model_id": "mistralai/voxtral-small-24b-2507-stt", "model_name": "Mistral: Voxtral Small 24B 2507 STT", "context_length": 0, "pricing": { "prompt": "0.00005", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/voxtral-mini-3b-2507@deepinfra", "name": "Mistral: Voxtral Mini 3B 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 16.666700000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000166667" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/voxtral-mini-3b-2507", "canonicalSlug": "mistralai/voxtral-mini-3b-2507-20260813", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 0, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Voxtral Mini 3B 2507 is a speech and audio understanding model from Mistral AI. It is suited for transcription, translation, and compact audio processing workloads.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | mistralai/voxtral-mini-3b-2507-20260813", "model_id": "mistralai/voxtral-mini-3b-2507", "model_name": "Mistral: Voxtral Mini 3B 2507", "context_length": 0, "pricing": { "prompt": "0.0000166667", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance-seed/seedream-5-0-lite@seed", "name": "ByteDance Seed: Seedream 5.0 Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance-seed/seedream-5-0-lite", "canonicalSlug": "bytedance-seed/seedream-5-0-lite-20260812", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 0, "maxCompletionTokens": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "frequency_penalty", "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Media", "instruct_type": null }, "description": "Seedream 5.0 Lite is an image generation model from ByteDance Seed. It is suited for professional visual creation that benefits from web-connected retrieval, complex-prompt comprehension, visual references, and broad knowledge...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance-seed/seedream-5-0-lite-20260812", "model_id": "bytedance-seed/seedream-5-0-lite", "model_name": "ByteDance Seed: Seedream 5.0 Lite", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.00000838323353293413", "image_output": "0.00000838323353293413", "discount": 0 }, "provider_name": "Seed", "tag": "seed", "quantization": "unknown", "max_completion_tokens": 0, "max_prompt_tokens": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.7-flash@google", "name": "Google: Gemini 3.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "output": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "cacheWrite": [ { "amount": 0.0208333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000208333333333333" } ], "other": [ { "amount": 3.75e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000375" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.7-flash", "canonicalSlug": "google/gemini-3.7-flash-20260813", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...", "uptimeLast30m": 98.06014618566866, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.7-flash-20260813", "model_id": "google/gemini-3.7-flash", "model_name": "Google: Gemini 3.7 Flash", "context_length": 1048576, "pricing": { "prompt": "0.000000375", "completion": "0.000001875", "image": "0.000000375", "audio": "0.000000375", "input_audio_cache": "0.0000000375", "web_search": "0.014", "internal_reasoning": "0.000001875", "input_cache_read": "0.0000000375", "input_cache_write": "0.0000000208333333333333", "discount": 0.75 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.06014618566866, "uptime_last_5m": 98.54962411724505, "uptime_last_1d": 98.88792120502248, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.7-flash@google", "name": "Google: Gemini 3.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "output": [ { "amount": 0.9375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009375" } ], "cacheRead": [ { "amount": 0.01875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001875" } ], "cacheWrite": [ { "amount": 0.0104166666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000104166666666667" } ], "other": [ { "amount": 1.875e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000001875" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.9375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009375" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.7-flash", "canonicalSlug": "google/gemini-3.7-flash-20260813", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...", "uptimeLast30m": 98.06014618566866, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.7-flash-20260813", "model_id": "google/gemini-3.7-flash", "model_name": "Google: Gemini 3.7 Flash", "context_length": 1048576, "pricing": { "prompt": "0.0000001875", "completion": "0.0000009375", "image": "0.0000001875", "audio": "0.0000001875", "input_audio_cache": "0.00000001875", "web_search": "0.014", "internal_reasoning": "0.0000009375", "input_cache_read": "0.00000001875", "input_cache_write": "0.0000000104166666666667", "discount": 0.75 }, "provider_name": "Google", "tag": "google-vertex/global/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.06014618566866, "uptime_last_5m": 98.54962411724505, "uptime_last_1d": 98.88792120502248, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.7-flash@google", "name": "Google: Gemini 3.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.675, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000675" } ], "output": [ { "amount": 3.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003375" } ], "cacheRead": [ { "amount": 0.0675, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000675" } ], "cacheWrite": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "other": [ { "amount": 6.75e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000675" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 3.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003375" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.7-flash", "canonicalSlug": "google/gemini-3.7-flash-20260813", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...", "uptimeLast30m": 98.06014618566866, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.7-flash-20260813", "model_id": "google/gemini-3.7-flash", "model_name": "Google: Gemini 3.7 Flash", "context_length": 1048576, "pricing": { "prompt": "0.000000675", "completion": "0.000003375", "image": "0.000000675", "audio": "0.000000675", "input_audio_cache": "0.0000000675", "web_search": "0.014", "internal_reasoning": "0.000003375", "input_cache_read": "0.0000000675", "input_cache_write": "0.0000000375", "discount": 0.75 }, "provider_name": "Google", "tag": "google-vertex/global/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.06014618566866, "uptime_last_5m": 98.54962411724505, "uptime_last_1d": 98.88792120502248, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.7-flash@google-ai-studio", "name": "Google: Gemini 3.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 7.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000075" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.7-flash", "canonicalSlug": "google/gemini-3.7-flash-20260813", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...", "uptimeLast30m": 99.92145629273702, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.7-flash-20260813", "model_id": "google/gemini-3.7-flash", "model_name": "Google: Gemini 3.7 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000075", "completion": "0.00000375", "image": "0.00000075", "audio": "0.00000075", "input_audio_cache": "0.000000075", "web_search": "0.014", "internal_reasoning": "0.00000375", "input_cache_read": "0.000000075", "input_cache_write": "0.0000000416666666666667", "discount": 0.5 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.92145629273702, "uptime_last_5m": 100, "uptime_last_1d": 99.80729883757687, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.7-flash@google-ai-studio", "name": "Google: Gemini 3.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "output": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "cacheWrite": [ { "amount": 0.0208333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000208333333333333" } ], "other": [ { "amount": 3.75e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000375" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.7-flash", "canonicalSlug": "google/gemini-3.7-flash-20260813", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...", "uptimeLast30m": 99.92145629273702, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.7-flash-20260813", "model_id": "google/gemini-3.7-flash", "model_name": "Google: Gemini 3.7 Flash", "context_length": 1048576, "pricing": { "prompt": "0.000000375", "completion": "0.000001875", "image": "0.000000375", "audio": "0.000000375", "input_audio_cache": "0.0000000375", "web_search": "0.014", "internal_reasoning": "0.000001875", "input_cache_read": "0.0000000375", "input_cache_write": "0.0000000208333333333333", "discount": 0.5 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.92145629273702, "uptime_last_5m": 100, "uptime_last_1d": 99.80729883757687, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.7-flash@google-ai-studio", "name": "Google: Gemini 3.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 6.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000675" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000135" } ], "cacheWrite": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "other": [ { "amount": 0.00000135, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000135" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 6.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000675" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.7-flash", "canonicalSlug": "google/gemini-3.7-flash-20260813", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...", "uptimeLast30m": 99.92145629273702, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.7-flash-20260813", "model_id": "google/gemini-3.7-flash", "model_name": "Google: Gemini 3.7 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000135", "completion": "0.00000675", "image": "0.00000135", "audio": "0.00000135", "input_audio_cache": "0.000000135", "web_search": "0.014", "internal_reasoning": "0.00000675", "input_cache_read": "0.000000135", "input_cache_write": "0.000000075", "discount": 0.5 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.92145629273702, "uptime_last_5m": 100, "uptime_last_1d": 99.80729883757687, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.7-flash:batch@google", "name": "Google: Gemini 3.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "output": [ { "amount": 0.9375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009375" } ], "cacheRead": [ { "amount": 0.01875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001875" } ], "cacheWrite": [ { "amount": 0.0208333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000208333333333333" } ], "other": [ { "amount": 1.875e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000001875" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.9375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009375" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.7-flash:batch", "canonicalSlug": "google/gemini-3.7-flash-20260813", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.7-flash-20260813:batch", "model_id": "google/gemini-3.7-flash:batch", "model_name": "Google: Gemini 3.7 Flash", "context_length": 1048576, "pricing": { "prompt": "0.0000001875", "completion": "0.0000009375", "image": "0.0000001875", "audio": "0.0000001875", "input_audio_cache": "0.00000001875", "web_search": "0.014", "internal_reasoning": "0.0000009375", "input_cache_read": "0.00000001875", "input_cache_write": "0.0000000208333333333333", "discount": 0.75 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "voyageai/voyage-code-4@voyageai-by-mongodb", "name": "VoyageAI by MongoDB: voyage-code-4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000012" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "voyageai/voyage-code-4", "canonicalSlug": "voyageai/voyage-code-4-20260812", "servingProvider": "VoyageAI by MongoDB", "servingProviderSlug": "voyageai-by-mongodb", "contextLength": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "voyage-code-4 is a code embedding model from Voyage AI, a MongoDB company. It is designed for coding agents and code retrieval, with Matryoshka embeddings at 2048, 1024, 512, and 256...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "VoyageAI by MongoDB | voyageai/voyage-code-4-20260812", "model_id": "voyageai/voyage-code-4", "model_name": "VoyageAI by MongoDB: voyage-code-4", "context_length": 32000, "pricing": { "prompt": "0.00000012", "completion": "0", "discount": 0 }, "provider_name": "VoyageAI by MongoDB", "tag": "voyageai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-reranker-8b@fireworks", "name": "Qwen3 Reranker 8B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-reranker-8b", "canonicalSlug": "qwen/qwen3-reranker-8b", "servingProvider": "Fireworks", "servingProviderSlug": "fireworks", "contextLength": 40960, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs" ], "architecture": { "modality": "text->rerank", "input_modalities": [ "text" ], "output_modalities": [ "rerank" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3 Reranker 8B is a text reranking model from Alibaba Cloud built on the Qwen3 architecture. It evaluates query-document pairs to produce relevance scores for use in retrieval and RAG...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fireworks | qwen/qwen3-reranker-8b", "model_id": "qwen/qwen3-reranker-8b", "model_name": "Qwen3 Reranker 8B", "context_length": 40960, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Fireworks", "tag": "fireworks", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-asr-1.7b@deepinfra", "name": "Qwen: Qwen3 ASR 1.7B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000075" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-asr-1.7b", "canonicalSlug": "qwen/qwen3-asr-1.7b-20260813", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3 ASR 1.7B is an automatic speech recognition model from Qwen. It supports multilingual language identification and transcription across 30 languages and 22 Chinese dialects, with streaming and offline inference...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-asr-1.7b-20260813", "model_id": "qwen/qwen3-asr-1.7b", "model_name": "Qwen: Qwen3 ASR 1.7B", "context_length": 0, "pricing": { "prompt": "0.0000075", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-asr-0.6b@deepinfra", "name": "Qwen: Qwen3 ASR 0.6B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000333" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-asr-0.6b", "canonicalSlug": "qwen/qwen3-asr-0.6b-20260813", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3 ASR 0.6B is a compact automatic speech recognition model from Qwen. It supports multilingual language identification and transcription across 30 languages and 22 Chinese dialects, with streaming and offline...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-asr-0.6b-20260813", "model_id": "qwen/qwen3-asr-0.6b", "model_name": "Qwen: Qwen3 ASR 0.6B", "context_length": 0, "pricing": { "prompt": "0.00000333", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance-seed/seedream-5-0-pro@seed", "name": "ByteDance Seed: Seedream 5.0 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.003, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.003" } ] } ], "metadata": { "source": "openrouter", "modelId": "bytedance-seed/seedream-5-0-pro", "canonicalSlug": "bytedance-seed/seedream-5-0-pro-20260812", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 0, "maxCompletionTokens": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "frequency_penalty", "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Media", "instruct_type": null }, "description": "Seedream 5.0 Pro is an image generation and editing model from ByteDance Seed. It is suited for commercial visual-production workflows that require precise editing control, lifelike scenes, and natural rendering.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance-seed/seedream-5-0-pro-20260812", "model_id": "bytedance-seed/seedream-5-0-pro", "model_name": "ByteDance Seed: Seedream 5.0 Pro", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "image": "0.003", "image_token": "0.0000107784431137725", "image_output": "0.0000107784431137725", "discount": 0 }, "provider_name": "Seed", "tag": "seed", "quantization": "unknown", "max_completion_tokens": 0, "max_prompt_tokens": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepgram/flux-tts:free@deepgram", "name": "Deepgram: Flux TTS", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepgram/flux-tts:free", "canonicalSlug": "deepgram/flux-tts-20260812", "servingProvider": "Deepgram", "servingProviderSlug": "deepgram", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "Flux TTS is a text-to-speech model from Deepgram. It is suited for natural, expressive English speech synthesis across Deepgram's Flux voice catalog.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Deepgram | deepgram/flux-tts-20260812:free", "model_id": "deepgram/flux-tts:free", "model_name": "Deepgram: Flux TTS", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Deepgram", "tag": "deepgram", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance/seedance-2.0-mini@seed", "name": "ByteDance: Seedance 2.0 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance/seedance-2.0-mini", "canonicalSlug": "bytedance/seedance-2.0-mini-20260811", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "frequency_penalty" ], "architecture": { "modality": "text+image+audio+video->video", "input_modalities": [ "text", "image", "video", "audio" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Seedance 2.0 Mini is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video with image, video, and audio inputs. It...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance/seedance-2.0-mini-20260811", "model_id": "bytedance/seedance-2.0-mini", "model_name": "ByteDance: Seedance 2.0 Mini", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0.6 }, "provider_name": "Seed", "tag": "seed", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "frequency_penalty" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance-seed/seed-2-1-turbo@seed", "name": "ByteDance Seed: Seed 2.1 Turbo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance-seed/seed-2-1-turbo", "canonicalSlug": "bytedance-seed/seed-2-1-turbo-20260810", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "stop", "tools", "response_format", "structured_outputs", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Seed 2.1 Turbo is a multimodal model from ByteDance Seed for coding and long-horizon agent workflows. It is suited for end-to-end software delivery, multi-step task execution, and understanding visual and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance-seed/seed-2-1-turbo-20260810", "model_id": "bytedance-seed/seed-2-1-turbo", "model_name": "ByteDance Seed: Seed 2.1 Turbo", "context_length": 262144, "pricing": { "prompt": "0.0000005", "completion": "0.0000025", "discount": 0 }, "provider_name": "Seed", "tag": "seed/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "stop", "tools", "response_format", "structured_outputs", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-2.4t-a95b@digitalocean", "name": "Qwen: Qwen3.8 2.4T A95B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-2.4t-a95b", "canonicalSlug": "qwen/qwen3.8-2.4t-a95b-20260812", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 262144, "maxCompletionTokens": 52429, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "tools", "tool_choice", "structured_outputs", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | qwen/qwen3.8-2.4t-a95b-20260812", "model_id": "qwen/qwen3.8-2.4t-a95b", "model_name": "Qwen: Qwen3.8 2.4T A95B", "context_length": 262144, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": 52429, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "tools", "tool_choice", "structured_outputs", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.86092244000258, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-2.4t-a95b@deepinfra", "name": "Qwen: Qwen3.8 2.4T A95B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-2.4t-a95b", "canonicalSlug": "qwen/qwen3.8-2.4t-a95b-20260812", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "fp4", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...", "uptimeLast30m": 89.375, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3.8-2.4t-a95b-20260812", "model_id": "qwen/qwen3.8-2.4t-a95b", "model_name": "Qwen: Qwen3.8 2.4T A95B", "context_length": 262144, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": -2, "uptime_last_30m": 89.375, "uptime_last_5m": null, "uptime_last_1d": 95.73149132128941, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-2.4t-a95b@modal", "name": "Qwen: Qwen3.8 2.4T A95B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-2.4t-a95b", "canonicalSlug": "qwen/qwen3.8-2.4t-a95b-20260812", "servingProvider": "Modal", "servingProviderSlug": "modal", "contextLength": 1000000, "maxCompletionTokens": 262144, "quantization": "nvfp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "stop", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Modal | qwen/qwen3.8-2.4t-a95b-20260812", "model_id": "qwen/qwen3.8-2.4t-a95b", "model_name": "Qwen: Qwen3.8 2.4T A95B", "context_length": 1000000, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "Modal", "tag": "modal/nvfp4", "quantization": "nvfp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "stop", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.3673050615595, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-2.4t-a95b@siliconflow", "name": "Qwen: Qwen3.8 2.4T A95B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-2.4t-a95b", "canonicalSlug": "qwen/qwen3.8-2.4t-a95b-20260812", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3.8-2.4t-a95b-20260812", "model_id": "qwen/qwen3.8-2.4t-a95b", "model_name": "Qwen: Qwen3.8 2.4T A95B", "context_length": 1048576, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.67725496857483, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-2.4t-a95b@alibaba", "name": "Qwen: Qwen3.8 2.4T A95B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-2.4t-a95b", "canonicalSlug": "qwen/qwen3.8-2.4t-a95b-20260812", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "top_k", "frequency_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.8-2.4t-a95b-20260812", "model_id": "qwen/qwen3.8-2.4t-a95b", "model_name": "Qwen: Qwen3.8 2.4T A95B", "context_length": 1000000, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.00000025", "input_cache_write": "0.0000025", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "top_k", "frequency_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.94295493439817, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-2.4t-a95b@together", "name": "Qwen: Qwen3.8 2.4T A95B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-2.4t-a95b", "canonicalSlug": "qwen/qwen3.8-2.4t-a95b-20260812", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 1010000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | qwen/qwen3.8-2.4t-a95b-20260812", "model_id": "qwen/qwen3.8-2.4t-a95b", "model_name": "Qwen: Qwen3.8 2.4T A95B", "context_length": 1010000, "pricing": { "prompt": "0.0000025", "completion": "0.00000625", "input_cache_read": "0.0000005", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.54995499549955, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-2.4t-a95b@venice", "name": "Qwen: Qwen3.8 2.4T A95B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 0.3125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-2.4t-a95b", "canonicalSlug": "qwen/qwen3.8-2.4t-a95b-20260812", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | qwen/qwen3.8-2.4t-a95b-20260812", "model_id": "qwen/qwen3.8-2.4t-a95b", "model_name": "Qwen: Qwen3.8 2.4T A95B", "context_length": 262144, "pricing": { "prompt": "0.0000025", "completion": "0.0000075", "input_cache_read": "0.0000003125", "discount": 0 }, "provider_name": "Venice", "tag": "venice", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 94.95151169423845, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance-seed/seed-2.0-code@seed", "name": "ByteDance Seed: Seed-2.0-Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance-seed/seed-2.0-code", "canonicalSlug": "bytedance-seed/seed-2.0-code-20260730", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "structured_outputs", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Seed 2.0 Code is a model from ByteDance Seed optimized for agentic coding. It is suited for frontend development, multilingual programming tasks, and coding-agent workflows in tools such as Claude...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance-seed/seed-2.0-code-20260730", "model_id": "bytedance-seed/seed-2.0-code", "model_name": "ByteDance Seed: Seed-2.0-Code", "context_length": 262144, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "discount": 0, "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.000001", "completion": "0.000006" } ] }, "provider_name": "Seed", "tag": "seed/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "structured_outputs", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@deepseek", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000066" } ], "output": [ { "amount": 1.9800000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000198" } ], "cacheRead": [ { "amount": 0.022, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000022" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "DeepSeek", "servingProviderSlug": "deepseek", "contextLength": 1048576, "maxCompletionTokens": 384000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepSeek | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1048576, "pricing": { "prompt": "0.00000066", "completion": "0.00000198", "input_cache_read": "0.000000022", "discount": 0, "overrides": [ { "utc_start": 1000, "utc_end": 100, "prompt": "0.00000066", "completion": "0.00000198", "input_cache_read": "0.000000022" }, { "utc_start": 100, "utc_end": 400, "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044" }, { "utc_start": 400, "utc_end": 600, "prompt": "0.00000066", "completion": "0.00000198", "input_cache_read": "0.000000022" }, { "utc_start": 600, "utc_end": 1000, "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044" } ] }, "provider_name": "DeepSeek", "tag": "deepseek", "quantization": "unknown", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.26550551518186, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@gmicloud", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.1880000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001188" } ], "output": [ { "amount": 3.564, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003564" } ], "cacheRead": [ { "amount": 0.039599999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000396" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 1048575, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 99.34309093004494, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1048575, "pricing": { "prompt": "0.000001188", "completion": "0.000003564", "input_cache_read": "0.0000000396", "discount": 0.1 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.34309093004494, "uptime_last_5m": 99.83870967741936, "uptime_last_1d": 99.70257411558524, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@baseten", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.13199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000132" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "BaseTen", "servingProviderSlug": "baseten", "contextLength": 1048576, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 97.29119638826185, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "BaseTen | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1048576, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000132", "discount": 0 }, "provider_name": "BaseTen", "tag": "baseten/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.29119638826185, "uptime_last_5m": 100, "uptime_last_1d": 93.47985796498102, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@siliconflow", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000044" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1048576, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.00000044", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.55238404151801, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@alibaba", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.13199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000132" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 393216, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "top_k", "frequency_penalty", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 99.89842559674962, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1000000, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000132", "discount": 0, "overrides": [ { "utc_start": 0, "utc_end": 1400, "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000132" }, { "utc_start": 1400, "utc_end": 0, "prompt": "0.000000726", "completion": "0.000002178", "input_cache_read": "0.0000000726" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "top_k", "frequency_penalty", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.89842559674962, "uptime_last_5m": 100, "uptime_last_1d": 99.45218261582801, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@novita", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.13199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000132" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp8", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 91.64556962025317, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1048576, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000132", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "reasoning_effort" ], "status": -2, "uptime_last_30m": 91.64556962025317, "uptime_last_5m": 86.27450980392157, "uptime_last_1d": 96.79542303574877, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@streamlake", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.044, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000044" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 1024000, "maxCompletionTokens": 384000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 99.18367346938776, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1024000, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044", "discount": 0 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.18367346938776, "uptime_last_5m": 100, "uptime_last_1d": 97.78249132279213, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@parasail", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.044, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000044" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 99.07120743034056, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1048576, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.07120743034056, "uptime_last_5m": 98.94736842105263, "uptime_last_1d": 97.54436254436254, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@fireworks", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.044, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000044" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "Fireworks", "servingProviderSlug": "fireworks", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 99.35483870967742, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fireworks | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1048576, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044", "discount": 0 }, "provider_name": "Fireworks", "tag": "fireworks", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.35483870967742, "uptime_last_5m": 100, "uptime_last_1d": 98.94535827438045, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@together", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 99.80535279805352, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1048576, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.00000013", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.80535279805352, "uptime_last_5m": 100, "uptime_last_1d": 98.18332710900867, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@digitalocean", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.044, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000044" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 99.63636363636364, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1048576, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.63636363636364, "uptime_last_5m": null, "uptime_last_1d": 97.94223414578458, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro-0813@cloudflare", "name": "DeepSeek: DeepSeek V4 Pro 0813", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.044, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000044" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro-0813", "canonicalSlug": "deepseek/deepseek-v4-pro-20260813", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.", "uptimeLast30m": 99.77628635346755, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | deepseek/deepseek-v4-pro-20260813", "model_id": "deepseek/deepseek-v4-pro-0813", "model_name": "DeepSeek: DeepSeek V4 Pro 0813", "context_length": 1048576, "pricing": { "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.77628635346755, "uptime_last_5m": 97.36842105263158, "uptime_last_1d": 96.33976457477189, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.6@xai", "name": "SpaceXAI: Grok 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.6", "canonicalSlug": "x-ai/grok-4.6-20260810", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 500000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.", "uptimeLast30m": 99.98227055138585, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.6-20260810", "model_id": "x-ai/grok-4.6", "model_name": "SpaceXAI: Grok 4.6", "context_length": 500000, "pricing": { "prompt": "0.000002", "completion": "0.000006", "web_search": "0.005", "input_cache_read": "0.0000005", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000012", "input_cache_read": "0.000001" } ] }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.98227055138585, "uptime_last_5m": 100, "uptime_last_1d": 99.89076379427888, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.6@xai", "name": "SpaceXAI: Grok 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.6", "canonicalSlug": "x-ai/grok-4.6-20260810", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 500000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.", "uptimeLast30m": 99.98227055138585, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.6-20260810", "model_id": "x-ai/grok-4.6", "model_name": "SpaceXAI: Grok 4.6", "context_length": 500000, "pricing": { "prompt": "0.000004", "completion": "0.000012", "web_search": "0.005", "input_cache_read": "0.000001", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000008", "completion": "0.000024", "input_cache_read": "0.000002" } ] }, "provider_name": "xAI", "tag": "xai/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.98227055138585, "uptime_last_5m": 100, "uptime_last_1d": 99.89076379427888, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.6@xai", "name": "SpaceXAI: Grok 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.6", "canonicalSlug": "x-ai/grok-4.6-20260810", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 500000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.6-20260810", "model_id": "x-ai/grok-4.6", "model_name": "SpaceXAI: Grok 4.6", "context_length": 500000, "pricing": { "prompt": "0.000002", "completion": "0.000006", "web_search": "0.005", "input_cache_read": "0.0000005", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000012", "input_cache_read": "0.000001" } ] }, "provider_name": "xAI", "tag": "xai/zdr", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.95266432242359, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.6@xai", "name": "SpaceXAI: Grok 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.6", "canonicalSlug": "x-ai/grok-4.6-20260810", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 500000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.6-20260810", "model_id": "x-ai/grok-4.6", "model_name": "SpaceXAI: Grok 4.6", "context_length": 500000, "pricing": { "prompt": "0.000004", "completion": "0.000012", "web_search": "0.005", "input_cache_read": "0.000001", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000008", "completion": "0.000024", "input_cache_read": "0.000002" } ] }, "provider_name": "xAI", "tag": "xai/zdr/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.95266432242359, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-imagine-image-2.0@xai", "name": "xAI: Grok Imagine Image 2.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-imagine-image-2.0", "canonicalSlug": "x-ai/grok-imagine-image-2.0-20260811", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok Imagine Image 2.0 is an image generation and editing model from xAI. It is suited for creating images from text prompts and editing images from references, with low and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-imagine-image-2.0-20260811", "model_id": "x-ai/grok-imagine-image-2.0", "model_name": "xAI: Grok Imagine Image 2.0", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image": "0.01", "image_token": "0.00000958083832335329", "image_output": "0.00000958083832335329", "discount": 0 }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.9677731227844, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "liquid/lfm-2.5-2.6b:free@liquid", "name": "LiquidAI: LFM2.5-2.6B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "liquid/lfm-2.5-2.6b:free", "canonicalSlug": "liquid/lfm-2.5-2.6b-20260811", "servingProvider": "Liquid", "servingProviderSlug": "liquid", "contextLength": 128000, "maxCompletionTokens": 8192, "quantization": "fp8", "status": -5, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "max_completion_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "LFM2.5-2.6B is a compact reasoning model from Liquid AI. It is suited for agent workflows, data extraction, RAG, and long-context processing. Liquid advises against using it for agentic coding or...", "uptimeLast30m": 76.64165103189492, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Liquid | liquid/lfm-2.5-2.6b-20260811:free", "model_id": "liquid/lfm-2.5-2.6b:free", "model_name": "LiquidAI: LFM2.5-2.6B", "context_length": 128000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Liquid", "tag": "liquid/fp8", "quantization": "fp8", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "max_completion_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": -5, "uptime_last_30m": 76.64165103189492, "uptime_last_5m": 82.01634877384197, "uptime_last_1d": 93.6681165150916, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3.5-lightning@deepinfra", "name": "NVIDIA: Nemotron 3.5 Lightning", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3.5-lightning", "canonicalSlug": "nvidia/nemotron-3.5-lightning-20260807", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | nvidia/nemotron-3.5-lightning-20260807", "model_id": "nvidia/nemotron-3.5-lightning", "model_name": "NVIDIA: Nemotron 3.5 Lightning", "context_length": 262144, "pricing": { "prompt": "0.00000008", "completion": "0.0000002", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.91882899677992, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3.5-lightning@coreweave", "name": "NVIDIA: Nemotron 3.5 Lightning", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3.5-lightning", "canonicalSlug": "nvidia/nemotron-3.5-lightning-20260807", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | nvidia/nemotron-3.5-lightning-20260807", "model_id": "nvidia/nemotron-3.5-lightning", "model_name": "NVIDIA: Nemotron 3.5 Lightning", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.00000025", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/bf16", "quantization": "bf16", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.87475555360243, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3.5-lightning@venice", "name": "NVIDIA: Nemotron 3.5 Lightning", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3.5-lightning", "canonicalSlug": "nvidia/nemotron-3.5-lightning-20260807", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 1000000, "maxCompletionTokens": 32768, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | nvidia/nemotron-3.5-lightning-20260807", "model_id": "nvidia/nemotron-3.5-lightning", "model_name": "NVIDIA: Nemotron 3.5 Lightning", "context_length": 1000000, "pricing": { "prompt": "0.0000001", "completion": "0.00000025", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp4", "quantization": "fp4", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.87986322890676, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3.5-lightning:free@nvidia", "name": "NVIDIA: Nemotron 3.5 Lightning", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3.5-lightning:free", "canonicalSlug": "nvidia/nemotron-3.5-lightning-20260807", "servingProvider": "Nvidia", "servingProviderSlug": "nvidia", "contextLength": 1000000, "maxCompletionTokens": 65536, "quantization": "nvfp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...", "uptimeLast30m": 99.75930907546369, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nvidia | nvidia/nemotron-3.5-lightning-20260807:free", "model_id": "nvidia/nemotron-3.5-lightning:free", "model_name": "NVIDIA: Nemotron 3.5 Lightning", "context_length": 1000000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Nvidia", "tag": "nvidia/nvfp4", "quantization": "nvfp4", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.75930907546369, "uptime_last_5m": 99.80940279542567, "uptime_last_1d": 99.79344300430705, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sakana/sakana-namazu@sakana-ai", "name": "Sakana: Sakana Namazu", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [ { "amount": 0.007, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.007" } ] } ], "metadata": { "source": "openrouter", "modelId": "sakana/sakana-namazu", "canonicalSlug": "sakana/namazu-20260811", "servingProvider": "Sakana AI", "servingProviderSlug": "sakana-ai", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "web_search_options", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Sakana Namazu is a Japanese-specialized reasoning model from Sakana AI, based on Kimi K2.6 with additional training for Japanese language and business contexts. It is suited for Japanese instruction following,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Sakana AI | sakana/namazu-20260811", "model_id": "sakana/sakana-namazu", "model_name": "Sakana: Sakana Namazu", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "web_search": "0.007", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "Sakana AI", "tag": "sakana", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "web_search_options", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "upstage/solar-pro4@upstage", "name": "Upstage: Solar Pro 4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "output": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "cacheRead": [ { "amount": 0.006, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "upstage/solar-pro4", "canonicalSlug": "upstage/solar-pro4-20260810", "servingProvider": "Upstage", "servingProviderSlug": "upstage", "contextLength": 524288, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Solar Pro 4 is Upstage's cost-efficient large language model, featuring a 524K context window. It is built for long-horizon tasks and agentic workflows, with strong capabilities in office productivity, document-intensive...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Upstage | upstage/solar-pro4-20260810", "model_id": "upstage/solar-pro4", "model_name": "Upstage: Solar Pro 4", "context_length": 524288, "pricing": { "prompt": "0.00000003", "completion": "0.00000012", "input_cache_read": "0.000000006", "discount": 0.9 }, "provider_name": "Upstage", "tag": "upstage", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99923608626624, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta/muse-glimmer-30b@parasail", "name": "Meta: Muse Glimmer 30B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta/muse-glimmer-30b", "canonicalSlug": "meta/muse-glimmer-30b-20260810", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | meta/muse-glimmer-30b-20260810", "model_id": "meta/muse-glimmer-30b", "model_name": "Meta: Muse Glimmer 30B", "context_length": 131072, "pricing": { "prompt": "0.0000003", "completion": "0.0000011", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta/muse-glimmer-30b@deepinfra", "name": "Meta: Muse Glimmer 30B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta/muse-glimmer-30b", "canonicalSlug": "meta/muse-glimmer-30b-20260810", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | meta/muse-glimmer-30b-20260810", "model_id": "meta/muse-glimmer-30b", "model_name": "Meta: Muse Glimmer 30B", "context_length": 131072, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.72565646488759, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta/muse-glimmer-30b@phala", "name": "Meta: Muse Glimmer 30B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta/muse-glimmer-30b", "canonicalSlug": "meta/muse-glimmer-30b-20260810", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "structured_outputs", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...", "uptimeLast30m": 99.50248756218906, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | meta/muse-glimmer-30b-20260810", "model_id": "meta/muse-glimmer-30b", "model_name": "Meta: Muse Glimmer 30B", "context_length": 131072, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "structured_outputs", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.50248756218906, "uptime_last_5m": 100, "uptime_last_1d": 99.83870110073205, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta/muse-glimmer-30b@together", "name": "Meta: Muse Glimmer 30B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta/muse-glimmer-30b", "canonicalSlug": "meta/muse-glimmer-30b-20260810", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | meta/muse-glimmer-30b-20260810", "model_id": "meta/muse-glimmer-30b", "model_name": "Meta: Muse Glimmer 30B", "context_length": 131072, "pricing": { "prompt": "0.00000035", "completion": "0.0000015", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.93604221213998, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta/muse-glimmer-30b@fireworks", "name": "Meta: Muse Glimmer 30B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta/muse-glimmer-30b", "canonicalSlug": "meta/muse-glimmer-30b-20260810", "servingProvider": "Fireworks", "servingProviderSlug": "fireworks", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fireworks | meta/muse-glimmer-30b-20260810", "model_id": "meta/muse-glimmer-30b", "model_name": "Meta: Muse Glimmer 30B", "context_length": 131072, "pricing": { "prompt": "0.00000035", "completion": "0.0000015", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "Fireworks", "tag": "fireworks", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.9650447427293, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance/seedance-2.5@seed", "name": "ByteDance: Seedance 2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance/seedance-2.5", "canonicalSlug": "bytedance/seedance-2.5-20260807", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "frequency_penalty" ], "architecture": { "modality": "text+image+audio+video->video", "input_modalities": [ "text", "image", "video", "audio" ], "output_modalities": [ "video" ], "tokenizer": "Media", "instruct_type": null }, "description": "Seedance 2.5 is a video generation model from ByteDance. It is suited for long-form storytelling, multimodal reference-based generation, video editing, and video extension. It supports first-frame and first-and-last-frame control, up...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance/seedance-2.5-20260807", "model_id": "bytedance/seedance-2.5", "model_name": "ByteDance: Seedance 2.5", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Seed", "tag": "seed", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "frequency_penalty" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-transcribe@openai", "name": "OpenAI: GPT Transcribe", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 4500, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0045" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-transcribe", "canonicalSlug": "openai/gpt-transcribe-20260805", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT Transcribe is a high-accuracy speech-to-text model from OpenAI. It is suited for recorded audio, streamed file transcription, and committed Realtime turns, with free-form context, keyword hints, and multiple language...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-transcribe-20260805", "model_id": "openai/gpt-transcribe", "model_name": "OpenAI: GPT Transcribe", "context_length": 0, "pricing": { "prompt": "0.0045", "completion": "0", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta/muse-spark-1.2@meta", "name": "Meta: Muse Spark 1.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 4.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000425" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [ { "amount": 0.0025, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.0025" } ] } ], "metadata": { "source": "openrouter", "modelId": "meta/muse-spark-1.2", "canonicalSlug": "meta/muse-spark-1.2-20260805", "servingProvider": "Meta", "servingProviderSlug": "meta", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "repetition_penalty", "top_k", "temperature", "top_p", "tools", "tool_choice", "structured_outputs", "response_format", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Meta | meta/muse-spark-1.2-20260805", "model_id": "meta/muse-spark-1.2", "model_name": "Meta: Muse Spark 1.2", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00000425", "web_search": "0.0025", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "Meta", "tag": "meta", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "repetition_penalty", "top_k", "temperature", "top_p", "tools", "tool_choice", "structured_outputs", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.97458727857989, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-image-3@alibaba", "name": "Qwen: Qwen Image 3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.003, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.003" } ] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-image-3", "canonicalSlug": "qwen/qwen-image-3-20260805", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen Image 3 is a unified image generation and editing model from Qwen. It supports precise rendering of text and details as small as 10px, along with a richer world...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen-image-3-20260805", "model_id": "qwen/qwen-image-3", "model_name": "Qwen: Qwen Image 3", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image": "0.003", "image_token": "0.00000718562874251497", "image_output": "0.00000718562874251497", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-image-3-pro@alibaba", "name": "Qwen: Qwen Image 3 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.003, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.003" } ] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-image-3-pro", "canonicalSlug": "qwen/qwen-image-3-pro-20260805", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen Image 3 Pro is an image generation and editing model from Qwen. It supports precise rendering of text and details as small as 10px, along with richer world knowledge...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen-image-3-pro-20260805", "model_id": "qwen/qwen-image-3-pro", "model_name": "Qwen: Qwen Image 3 Pro", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image": "0.003", "image_token": "0.00000958083832335329", "image_output": "0.00000958083832335329", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "black-forest-labs/flux-3-video@black-forest-labs", "name": "Black Forest Labs: FLUX.3 Video", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "black-forest-labs/flux-3-video", "canonicalSlug": "black-forest-labs/flux-3-video-20260804", "servingProvider": "Black Forest Labs", "servingProviderSlug": "black-forest-labs", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image+video->video", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "video" ], "tokenizer": "Media", "instruct_type": null }, "description": "FLUX.3 Video is a video generation model from Black Forest Labs. It supports text-to-video, image-guided generation with opening and closing keyframes, and video continuation workflows, making it suited for controlled...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Black Forest Labs | black-forest-labs/flux-3-video-20260804", "model_id": "black-forest-labs/flux-3-video", "model_name": "Black Forest Labs: FLUX.3 Video", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Black Forest Labs", "tag": "black-forest-labs", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.8-max@alibaba", "name": "Qwen: Qwen3.8 Max", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.8-max", "canonicalSlug": "qwen/qwen3.8-max-20260803", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.8 Max is the flagship model in Alibaba's Qwen3.8 series, the general-availability successor to the Qwen3.8 Max Preview. It is a multimodal reasoning model intended for complex reasoning, visual understanding,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.8-max-20260803", "model_id": "qwen/qwen3.8-max", "model_name": "Qwen: Qwen3.8 Max", "context_length": 1000000, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.00000025", "input_cache_write": "0.0000025", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": 983616, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99987707512553, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "~deepseek/deepseek-v4-flash-latest", "name": "DeepSeek V4 Flash Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.0765, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000765" } ], "output": [ { "amount": 0.153, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000153" } ], "cacheRead": [ { "amount": 0.015300000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000153" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "canonicalSlug": "~deepseek/deepseek-v4-flash-latest", "contextLength": 1310720, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "parallel_tool_calls", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_a", "top_k", "top_logprobs", "top_p" ], "description": "This model always redirects to the latest model in the DeepSeek V4 Flash family.", "endpointCount": 0 } }, { "id": "deepseek/deepseek-v4-flash-0731@decart", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.0765, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000765" } ], "output": [ { "amount": 0.153, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000153" } ], "cacheRead": [ { "amount": 0.015300000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000153" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Decart", "servingProviderSlug": "decart", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": -5, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 34.38143666323378, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Decart | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 262144, "pricing": { "prompt": "0.0000000765", "completion": "0.000000153", "input_cache_read": "0.0000000153", "discount": 0.15 }, "provider_name": "Decart", "tag": "decart/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": -5, "uptime_last_30m": 34.38143666323378, "uptime_last_5m": 9.246231155778894, "uptime_last_1d": 80.1305507465727, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@streamlake", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.078596, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000078596" } ], "output": [ { "amount": 0.157192, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000157192" } ], "cacheRead": [ { "amount": 0.015719200000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000157192" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 1024000, "maxCompletionTokens": 384000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 98.12781076850449, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1024000, "pricing": { "prompt": "0.000000078596", "completion": "0.000000157192", "input_cache_read": "0.0000000157192", "discount": 0.4386 }, "provider_name": "StreamLake", "tag": "streamlake/fp8", "quantization": "fp8", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.12781076850449, "uptime_last_5m": 98.32221548413179, "uptime_last_1d": 95.86680209565031, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@digitalocean", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.079996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000079996" } ], "output": [ { "amount": 0.252, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000252" } ], "cacheRead": [ { "amount": 0.0252, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000252" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 96.5770663445082, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.000000079996", "completion": "0.000000252", "input_cache_read": "0.0000000252", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 96.5770663445082, "uptime_last_5m": 99.44036376355369, "uptime_last_1d": 98.42299790075484, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@deepinfra", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "cacheRead": [ { "amount": 0.016, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 1048576, "maxCompletionTokens": 384000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 99.12080738848175, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000008", "completion": "0.00000018", "input_cache_read": "0.000000016", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.12080738848175, "uptime_last_5m": 99.03757249726823, "uptime_last_1d": 99.03433917276965, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@openinference", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "OpenInference", "servingProviderSlug": "openinference", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "stop", "seed", "top_p", "frequency_penalty", "presence_penalty", "structured_outputs", "response_format", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 92.86906387498104, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenInference | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 262144, "pricing": { "prompt": "0.00000008", "completion": "0.00000018", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "OpenInference", "tag": "open-inference/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "stop", "seed", "top_p", "frequency_penalty", "presence_penalty", "structured_outputs", "response_format", "tools", "tool_choice", "reasoning_effort" ], "status": -2, "uptime_last_30m": 92.86906387498104, "uptime_last_5m": 95.69049951028403, "uptime_last_1d": 96.11172313015562, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@sail-research", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Sail Research", "servingProviderSlug": "sail-research", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 98.28252172713478, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Sail Research | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 262144, "pricing": { "prompt": "0.00000009", "completion": "0.00000018", "input_cache_read": "0.00000002", "discount": 0 }, "provider_name": "Sail Research", "tag": "sail-research/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.28252172713478, "uptime_last_5m": 98.36670179135932, "uptime_last_1d": 97.1546128681238, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@relace", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.105, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000105" } ], "output": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000021" } ], "cacheRead": [ { "amount": 0.020999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000021" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Relace", "servingProviderSlug": "relace", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "temperature", "top_p", "top_k", "min_p", "stop", "max_tokens", "logit_bias", "frequency_penalty", "presence_penalty", "repetition_penalty", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 99.31241942415127, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Relace | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.000000105", "completion": "0.00000021", "input_cache_read": "0.000000021", "discount": 0 }, "provider_name": "Relace", "tag": "relace/fp4", "quantization": "fp4", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "temperature", "top_p", "top_k", "min_p", "stop", "max_tokens", "logit_bias", "frequency_penalty", "presence_penalty", "repetition_penalty", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.31241942415127, "uptime_last_5m": 99.64564138908575, "uptime_last_1d": 96.85045012764813, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@gmicloud", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.112, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000112" } ], "output": [ { "amount": 0.224, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000224" } ], "cacheRead": [ { "amount": 0.0224, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000224" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 1048575, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 97.3777965386239, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048575, "pricing": { "prompt": "0.000000112", "completion": "0.000000224", "input_cache_read": "0.0000000224", "discount": 0.2 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.3777965386239, "uptime_last_5m": 99.81635824268965, "uptime_last_1d": 98.64145258397808, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@baseten", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "BaseTen", "servingProviderSlug": "baseten", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp8", "status": -5, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 61.8231046931408, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "BaseTen | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000013", "completion": "0.00000026", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "BaseTen", "tag": "baseten/fp8", "quantization": "fp8", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "status": -5, "uptime_last_30m": 61.8231046931408, "uptime_last_5m": 99.4186046511628, "uptime_last_1d": 98.53951622124511, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@coreweave", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 99.88421705200204, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 262144, "pricing": { "prompt": "0.00000013", "completion": "0.00000028", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.88421705200204, "uptime_last_5m": 99.96372869060573, "uptime_last_1d": 99.87521642061964, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@inceptron", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Inceptron", "servingProviderSlug": "inceptron", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "stop", "frequency_penalty", "logit_bias", "parallel_tool_calls", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 99.27853437094683, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Inceptron | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000013", "completion": "0.00000028", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Inceptron", "tag": "inceptron/fp4", "quantization": "fp4", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "stop", "frequency_penalty", "logit_bias", "parallel_tool_calls", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.27853437094683, "uptime_last_5m": 99.24471299093656, "uptime_last_1d": 98.34831889387809, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@morph", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13899999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000139" } ], "output": [ { "amount": 0.27799999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000278" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Morph", "servingProviderSlug": "morph", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 98.71043376318875, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Morph | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.000000139", "completion": "0.000000278", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "Morph", "tag": "morph/bf16", "quantization": "bf16", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.71043376318875, "uptime_last_5m": 99.38366718027734, "uptime_last_1d": 95.45698976096172, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@fireworks", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Fireworks", "servingProviderSlug": "fireworks", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 98.03872265526779, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fireworks | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "Fireworks", "tag": "fireworks", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.03872265526779, "uptime_last_5m": 99.35838680109991, "uptime_last_1d": 95.17071136090316, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@akashml", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "AkashML", "servingProviderSlug": "akashml", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 98.84935855045629, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AkashML | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 131072, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.00000002", "discount": 0 }, "provider_name": "AkashML", "tag": "akashml/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.84935855045629, "uptime_last_5m": 98.8984088127295, "uptime_last_1d": 95.26008911828372, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@novita", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "response_format", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 98.98350250007485, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "response_format", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.98350250007485, "uptime_last_5m": 99.98702478266512, "uptime_last_1d": 99.2326872076636, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@together", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 97.36900287419854, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.36900287419854, "uptime_last_5m": 97.55274261603375, "uptime_last_1d": 96.0429878859226, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@parasail", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 98.64864864864865, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.64864864864865, "uptime_last_5m": 98.22521419828641, "uptime_last_1d": 97.60741090091821, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@atlascloud", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 99.380404667954, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp4", "quantization": "fp4", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.380404667954, "uptime_last_5m": 99.41159586681975, "uptime_last_1d": 99.33426407498939, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@siliconflow", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 98.92185473411155, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.92185473411155, "uptime_last_5m": 99.29428369795342, "uptime_last_1d": 98.01708107937644, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@ambient", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Ambient", "servingProviderSlug": "ambient", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "frequency_penalty", "logit_bias", "min_p", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 98.74458393511833, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Ambient | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "Ambient", "tag": "ambient/fp4", "quantization": "fp4", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "frequency_penalty", "logit_bias", "min_p", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.74458393511833, "uptime_last_5m": 98.88412017167381, "uptime_last_1d": 95.11831502606528, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@baidu", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Baidu", "servingProviderSlug": "baidu", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 99.83125210934864, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Baidu | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "Baidu", "tag": "baidu/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.83125210934864, "uptime_last_5m": 99.77477477477478, "uptime_last_1d": 98.4273564136226, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@mancer-2", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.145, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000145" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Mancer 2", "servingProviderSlug": "mancer-2", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "response_format", "logprobs", "top_logprobs", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 96.68601838413159, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mancer 2 | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.000000145", "completion": "0.00000045", "discount": 0 }, "provider_name": "Mancer 2", "tag": "mancer/fp8", "quantization": "fp8", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "response_format", "logprobs", "top_logprobs", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 96.68601838413159, "uptime_last_5m": 96.93396226415094, "uptime_last_1d": 96.01305923692001, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@io-net", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.149, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000149" } ], "output": [ { "amount": 0.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000032" } ], "cacheRead": [ { "amount": 0.077, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000077" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Io Net", "servingProviderSlug": "io-net", "contextLength": 262100, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "seed", "tools", "tool_choice", "logprobs", "top_logprobs", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Io Net | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 262100, "pricing": { "prompt": "0.000000149", "completion": "0.00000032", "input_cache_read": "0.000000077", "discount": 0 }, "provider_name": "Io Net", "tag": "io-net/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "seed", "tools", "tool_choice", "logprobs", "top_logprobs", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 98.5891302206251, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@venice", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000175" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "cacheRead": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000035" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 1000000, "maxCompletionTokens": 32768, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 85.67014291232135, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1000000, "pricing": { "prompt": "0.000000175", "completion": "0.00000035", "input_cache_read": "0.000000035", "discount": 0 }, "provider_name": "Venice", "tag": "venice", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": -2, "uptime_last_30m": 85.67014291232135, "uptime_last_5m": 96.62288930581614, "uptime_last_1d": 94.39934265991637, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@phala", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 97.94268919911829, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.0000002", "completion": "0.0000004", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.94268919911829, "uptime_last_5m": 99.9178307313065, "uptime_last_1d": 97.29981556117221, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@deepseek", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000066" } ], "cacheRead": [ { "amount": 0.007, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "DeepSeek", "servingProviderSlug": "deepseek", "contextLength": 1048576, "maxCompletionTokens": 384000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 99.5086682883509, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepSeek | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000022", "completion": "0.00000066", "input_cache_read": "0.000000007", "discount": 0, "overrides": [ { "utc_start": 1000, "utc_end": 100, "prompt": "0.00000022", "completion": "0.00000066", "input_cache_read": "0.000000007" }, { "utc_start": 100, "utc_end": 400, "prompt": "0.00000044", "completion": "0.00000132", "input_cache_read": "0.000000014" }, { "utc_start": 400, "utc_end": 600, "prompt": "0.00000022", "completion": "0.00000066", "input_cache_read": "0.000000007" }, { "utc_start": 600, "utc_end": 1000, "prompt": "0.00000044", "completion": "0.00000132", "input_cache_read": "0.000000014" } ] }, "provider_name": "DeepSeek", "tag": "deepseek", "quantization": "unknown", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.5086682883509, "uptime_last_5m": 100, "uptime_last_1d": 96.89800360740982, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@wafer", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "output": [ { "amount": 0.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000056" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Wafer", "servingProviderSlug": "wafer", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 99.80061627696212, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Wafer | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1048576, "pricing": { "prompt": "0.00000028", "completion": "0.00000056", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "Wafer", "tag": "wafer/fast", "quantization": "unknown", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.80061627696212, "uptime_last_5m": 99.79423868312757, "uptime_last_1d": 98.85437971544707, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash-0731@cloudflare", "name": "DeepSeek: DeepSeek V4 Flash 0731", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000044" } ], "output": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "cacheRead": [ { "amount": 0.014, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000014" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash-0731", "canonicalSlug": "deepseek/deepseek-v4-flash-20260731", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 1310720, "maxCompletionTokens": 1310720, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....", "uptimeLast30m": 99.95046439628483, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | deepseek/deepseek-v4-flash-20260731", "model_id": "deepseek/deepseek-v4-flash-0731", "model_name": "DeepSeek: DeepSeek V4 Flash 0731", "context_length": 1310720, "pricing": { "prompt": "0.00000044", "completion": "0.00000132", "input_cache_read": "0.000000014", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 1310720, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.95046439628483, "uptime_last_5m": 99.84472049689441, "uptime_last_1d": 99.94627122286697, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thinkingmachines/inkling-small@deepinfra", "name": "Thinking Machines: Inkling Small", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thinkingmachines/inkling-small", "canonicalSlug": "thinkingmachines/inkling-small-20260730", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 524288, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+audio->text", "input_modalities": [ "text", "image", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...", "uptimeLast30m": 95.17523896222121, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | thinkingmachines/inkling-small-20260730", "model_id": "thinkingmachines/inkling-small", "model_name": "Thinking Machines: Inkling Small", "context_length": 524288, "pricing": { "prompt": "0.00000045", "completion": "0.0000012", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 95.17523896222121, "uptime_last_5m": 100, "uptime_last_1d": 99.36913278396571, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thinkingmachines/inkling-small@together", "name": "Thinking Machines: Inkling Small", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thinkingmachines/inkling-small", "canonicalSlug": "thinkingmachines/inkling-small-20260730", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 524288, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text+image+audio->text", "input_modalities": [ "text", "image", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...", "uptimeLast30m": 95.28089887640449, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | thinkingmachines/inkling-small-20260730", "model_id": "thinkingmachines/inkling-small", "model_name": "Thinking Machines: Inkling Small", "context_length": 524288, "pricing": { "prompt": "0.0000005", "completion": "0.0000012", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 95.28089887640449, "uptime_last_5m": 100, "uptime_last_1d": 99.7100790176795, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/hailuo-3@minimax", "name": "MiniMax: H3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/hailuo-3", "canonicalSlug": "minimax/hailuo-03-20260730", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text+image+audio+video->video", "input_modalities": [ "text", "image", "video", "audio" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax H3 is a lightweight, open-weights video generation model from MiniMax. It is designed for precise multimodal editing and controlled content generation, including instruction-guided edits, text and brand rendering, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/hailuo-03-20260730", "model_id": "minimax/hailuo-3", "model_name": "MiniMax: H3", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "fish-audio/transcribe-1@fish-audio", "name": "Fish Audio: Transcribe 1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 100, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "fish-audio/transcribe-1", "canonicalSlug": "fish-audio/transcribe-1-20260729", "servingProvider": "Fish Audio", "servingProviderSlug": "fish-audio", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Other", "instruct_type": null }, "description": "Transcribe 1 is a speech-to-text model from Fish Audio. It is suited for audio transcription with automatic language detection and can return timestamped word-level segments when alignment details are requested.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fish Audio | fish-audio/transcribe-1-20260729", "model_id": "fish-audio/transcribe-1", "model_name": "Fish Audio: Transcribe 1", "context_length": 0, "pricing": { "prompt": "0.0001", "completion": "0", "discount": 0 }, "provider_name": "Fish Audio", "tag": "fish-audio", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "fish-audio/s1@fish-audio", "name": "Fish Audio: S1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "fish-audio/s1", "canonicalSlug": "fish-audio/s1-20260729", "servingProvider": "Fish Audio", "servingProviderSlug": "fish-audio", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "S1 is a multilingual text-to-speech model from Fish Audio. It is suited for voice applications that need broad emotional expression, using parenthetical controls to guide speaking style across its supported...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fish Audio | fish-audio/s1-20260729", "model_id": "fish-audio/s1", "model_name": "Fish Audio: S1", "context_length": 0, "pricing": { "prompt": "0.000015", "completion": "0", "discount": 0 }, "provider_name": "Fish Audio", "tag": "fish-audio", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "fish-audio/s2-pro@fish-audio", "name": "Fish Audio: S2 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "fish-audio/s2-pro", "canonicalSlug": "fish-audio/s2-pro-20260729", "servingProvider": "Fish Audio", "servingProviderSlug": "fish-audio", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "S2 Pro is a multilingual text-to-speech model from Fish Audio. It is suited for expressive narration and multi-speaker dialogue, with natural-language controls for speaking style and emotion.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fish Audio | fish-audio/s2-pro-20260729", "model_id": "fish-audio/s2-pro", "model_name": "Fish Audio: S2 Pro", "context_length": 0, "pricing": { "prompt": "0.000015", "completion": "0", "discount": 0 }, "provider_name": "Fish Audio", "tag": "fish-audio", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "fish-audio/s2.1-pro-free:free@fish-audio", "name": "Fish Audio: S2.1 Pro Free", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "fish-audio/s2.1-pro-free:free", "canonicalSlug": "fish-audio/s2.1-pro-free-20260729", "servingProvider": "Fish Audio", "servingProviderSlug": "fish-audio", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "S2.1 Pro Free is the no-cost variant of Fish Audio S2.1 Pro, intended for testing, prototyping, and low-volume applications. It provides the same synthesis capabilities without production latency or availability...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fish Audio | fish-audio/s2.1-pro-free-20260729:free", "model_id": "fish-audio/s2.1-pro-free:free", "model_name": "Fish Audio: S2.1 Pro Free", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Fish Audio", "tag": "fish-audio", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "fish-audio/s2.1-pro@fish-audio", "name": "Fish Audio: S2.1 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "fish-audio/s2.1-pro", "canonicalSlug": "fish-audio/s2.1-pro-20260729", "servingProvider": "Fish Audio", "servingProviderSlug": "fish-audio", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "S2.1 Pro is a production-oriented text-to-speech model from Fish Audio. It is suited for multilingual voice applications, expressive narration, and dialogue synthesis, with open-ended natural-language controls for speaking style and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fish Audio | fish-audio/s2.1-pro-20260729", "model_id": "fish-audio/s2.1-pro", "model_name": "Fish Audio: S2.1 Pro", "context_length": 0, "pricing": { "prompt": "0.000015", "completion": "0", "discount": 0 }, "provider_name": "Fish Audio", "tag": "fish-audio", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": true, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "runway/aleph-2@runway", "name": "Runway: Aleph 2.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "runway/aleph-2", "canonicalSlug": "runway/aleph-2-20260729", "servingProvider": "Runway", "servingProviderSlug": "runway", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image+video->video", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "video" ], "tokenizer": "Media", "instruct_type": null }, "description": "Runway Aleph 2.0 is an in-context video editing model from Runway. It applies text instructions and keyframe-guided edits across existing footage while preserving details that are not meant to change....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Runway | runway/aleph-2-20260729", "model_id": "runway/aleph-2", "model_name": "Runway: Aleph 2.0", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Runway", "tag": "runway", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "runway/gen-4.5@runway", "name": "Runway: Gen-4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "runway/gen-4.5", "canonicalSlug": "runway/gen-4.5-20260729", "servingProvider": "Runway", "servingProviderSlug": "runway", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Media", "instruct_type": null }, "description": "Runway Gen-4.5 is a video generation model from Runway for text-to-video and image-to-video workflows. It is designed for cinematic scene creation with strong motion quality, visual fidelity, and prompt adherence....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Runway | runway/gen-4.5-20260729", "model_id": "runway/gen-4.5", "model_name": "Runway: Gen-4.5", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Runway", "tag": "runway", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.7-flash@alibaba", "name": "Qwen: Qwen3.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "output": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheRead": [ { "amount": 0.006, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000006" } ], "cacheWrite": [ { "amount": 0.038000000000000006, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000038" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.7-flash", "canonicalSlug": "qwen/qwen3.7-flash-20260727", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.7-flash-20260727", "model_id": "qwen/qwen3.7-flash", "model_name": "Qwen: Qwen3.7 Flash", "context_length": 1000000, "pricing": { "prompt": "0.00000003", "completion": "0.00000013", "input_cache_read": "0.000000006", "input_cache_write": "0.000000038", "discount": 0, "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.0000001", "completion": "0.0000004", "input_cache_read": "0.00000002", "input_cache_write": "0.000000125" }, { "min_prompt_tokens": 256000, "prompt": "0.0000002", "completion": "0.0000008", "input_cache_read": "0.00000004", "input_cache_write": "0.00000025" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 983616, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99949451753722, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "voyageai/rerank-2.5-lite@voyageai-by-mongodb", "name": "VoyageAI by MongoDB: rerank-2.5-lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "voyageai/rerank-2.5-lite", "canonicalSlug": "voyageai/rerank-2.5-lite-20260727", "servingProvider": "VoyageAI by MongoDB", "servingProviderSlug": "voyageai-by-mongodb", "contextLength": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->rerank", "input_modalities": [ "text" ], "output_modalities": [ "rerank" ], "tokenizer": "Other", "instruct_type": null }, "description": "rerank-2.5-lite is a reranker optimized for both latency and quality, delivering a 7.16% improvement in retrieval accuracy over Cohere Rerank v3.5 across 93 datasets. It also outperformed Cohere Rerank v3.5...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "VoyageAI by MongoDB | voyageai/rerank-2.5-lite-20260727", "model_id": "voyageai/rerank-2.5-lite", "model_name": "VoyageAI by MongoDB: rerank-2.5-lite", "context_length": 32000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "VoyageAI by MongoDB", "tag": "voyageai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "voyageai/rerank-2.5@voyageai-by-mongodb", "name": "VoyageAI by MongoDB: rerank-2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "voyageai/rerank-2.5", "canonicalSlug": "voyageai/rerank-2.5-20260727", "servingProvider": "VoyageAI by MongoDB", "servingProviderSlug": "voyageai-by-mongodb", "contextLength": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->rerank", "input_modalities": [ "text" ], "output_modalities": [ "rerank" ], "tokenizer": "Other", "instruct_type": null }, "description": "rerank-2.5 is a cutting-edge reranker optimized for quality, delivering a 7.94% improvement in retrieval accuracy over Cohere Rerank v3.5 across 93 datasets. It also outperformed Cohere Rerank v3.5 by 12.70%...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "VoyageAI by MongoDB | voyageai/rerank-2.5-20260727", "model_id": "voyageai/rerank-2.5", "model_name": "VoyageAI by MongoDB: rerank-2.5", "context_length": 32000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "VoyageAI by MongoDB", "tag": "voyageai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "voyageai/voyage-multimodal-3.5@voyageai-by-mongodb", "name": "VoyageAI by MongoDB: voyage-multimodal-3.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000012" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "voyageai/voyage-multimodal-3.5", "canonicalSlug": "voyageai/voyage-multimodal-3.5-20260727", "servingProvider": "VoyageAI by MongoDB", "servingProviderSlug": "voyageai-by-mongodb", "contextLength": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->embeddings", "input_modalities": [ "text", "image" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "voyage-multimodal-3.5 is a state-of-the-art multimodal embedding model capable of vectorizing not only text, images, and video individually, but also content that interleaves all three modalities. It delivers excellent performance for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "VoyageAI by MongoDB | voyageai/voyage-multimodal-3.5-20260727", "model_id": "voyageai/voyage-multimodal-3.5", "model_name": "VoyageAI by MongoDB: voyage-multimodal-3.5", "context_length": 32000, "pricing": { "prompt": "0.00000012", "completion": "0", "discount": 0 }, "provider_name": "VoyageAI by MongoDB", "tag": "voyageai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "voyageai/voyage-4-lite@voyageai-by-mongodb", "name": "VoyageAI by MongoDB: voyage-4-lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000002" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "voyageai/voyage-4-lite", "canonicalSlug": "voyageai/voyage-4-lite-20260727", "servingProvider": "VoyageAI by MongoDB", "servingProviderSlug": "voyageai-by-mongodb", "contextLength": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "voyage-4-lite is a lightweight, general-purpose embedding model optimized for low latency and cost. Enabled by Matryoshka learning and quantization-aware training, voyage-4-lite supports embeddings in 2048, 1024, 512, and 256 dimensions,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "VoyageAI by MongoDB | voyageai/voyage-4-lite-20260727", "model_id": "voyageai/voyage-4-lite", "model_name": "VoyageAI by MongoDB: voyage-4-lite", "context_length": 32000, "pricing": { "prompt": "0.00000002", "completion": "0", "discount": 0 }, "provider_name": "VoyageAI by MongoDB", "tag": "voyageai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "voyageai/voyage-4@voyageai-by-mongodb", "name": "VoyageAI by MongoDB: voyage-4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000006" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "voyageai/voyage-4", "canonicalSlug": "voyageai/voyage-4-20260727", "servingProvider": "VoyageAI by MongoDB", "servingProviderSlug": "voyageai-by-mongodb", "contextLength": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "voyage-4 is a general-purpose (including multilingual) embedding model optimized for retrieval/search and AI applications. voyage-4 supports embeddings in 2048, 1024, 512, and 256 dimensions, with multiple quantization options. Learn more...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "VoyageAI by MongoDB | voyageai/voyage-4-20260727", "model_id": "voyageai/voyage-4", "model_name": "VoyageAI by MongoDB: voyage-4", "context_length": 32000, "pricing": { "prompt": "0.00000006", "completion": "0", "discount": 0 }, "provider_name": "VoyageAI by MongoDB", "tag": "voyageai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "voyageai/voyage-4-large@voyageai-by-mongodb", "name": "VoyageAI by MongoDB: voyage-4-large", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000012" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "voyageai/voyage-4-large", "canonicalSlug": "voyageai/voyage-4-large-20260727", "servingProvider": "VoyageAI by MongoDB", "servingProviderSlug": "voyageai-by-mongodb", "contextLength": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "voyage-4-large is a state-of-the-art general-purpose and multilingual embedding optimized for retrieval quality. Enabled by Matryoshka learning and quantization-aware training, voyage-4-large supports embeddings in 2048, 1024, 512, and 256 dimensions, with...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "VoyageAI by MongoDB | voyageai/voyage-4-large-20260727", "model_id": "voyageai/voyage-4-large", "model_name": "VoyageAI by MongoDB: voyage-4-large", "context_length": 32000, "pricing": { "prompt": "0.00000012", "completion": "0", "discount": 0 }, "provider_name": "VoyageAI by MongoDB", "tag": "voyageai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-5-fast@anthropic", "name": "Claude Opus 5 (Fast)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-5-fast", "canonicalSlug": "anthropic/claude-opus-5-fast-20260723", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Fast-mode variant of [Opus 5](/anthropic/claude-opus-5) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 5.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-opus-5-fast-20260723", "model_id": "anthropic/claude-opus-5-fast", "model_name": "Claude Opus 5 (Fast)", "context_length": 1000000, "pricing": { "prompt": "0.00001", "completion": "0.00005", "web_search": "0.01", "input_cache_read": "0.000001", "input_cache_write": "0.0000125", "input_cache_write_1h": "0.00002", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-5@amazon-bedrock", "name": "Claude Opus 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-5", "canonicalSlug": "anthropic/claude-opus-5-20260723", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "uptimeLast30m": 99.91888822797816, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-opus-5-20260723", "model_id": "anthropic/claude-opus-5", "model_name": "Claude Opus 5", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.91888822797816, "uptime_last_5m": 99.91142604074402, "uptime_last_1d": 99.63577449854917, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-5@claude-platform-on-aws", "name": "Claude Opus 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-5", "canonicalSlug": "anthropic/claude-opus-5-20260723", "servingProvider": "Claude Platform on AWS", "servingProviderSlug": "claude-platform-on-aws", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "uptimeLast30m": 99.9083241657499, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Claude Platform on AWS | anthropic/claude-opus-5-20260723", "model_id": "anthropic/claude-opus-5", "model_name": "Claude Opus 5", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Claude Platform on AWS", "tag": "claude-on-aws", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.9083241657499, "uptime_last_5m": 100, "uptime_last_1d": 97.28435574856901, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-5@azure", "name": "Claude Opus 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-5", "canonicalSlug": "anthropic/claude-opus-5-20260723", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-opus-5-20260723", "model_id": "anthropic/claude-opus-5", "model_name": "Claude Opus 5", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.8757543485978, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-5@anthropic", "name": "Claude Opus 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-5", "canonicalSlug": "anthropic/claude-opus-5-20260723", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "uptimeLast30m": 99.94623655913979, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-opus-5-20260723", "model_id": "anthropic/claude-opus-5", "model_name": "Claude Opus 5", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.94623655913979, "uptime_last_5m": 100, "uptime_last_1d": 98.3540510993238, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-5@google", "name": "Claude Opus 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-5", "canonicalSlug": "anthropic/claude-opus-5-20260723", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "uptimeLast30m": 99.79674796747967, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-opus-5-20260723", "model_id": "anthropic/claude-opus-5", "model_name": "Claude Opus 5", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.79674796747967, "uptime_last_5m": 100, "uptime_last_1d": 99.47366322421149, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-5@azure", "name": "Claude Opus 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-5", "canonicalSlug": "anthropic/claude-opus-5-20260723", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-opus-5-20260723", "model_id": "anthropic/claude-opus-5", "model_name": "Claude Opus 5", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Azure", "tag": "azure/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.98894172287957, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-5@amazon-bedrock", "name": "Claude Opus 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-5", "canonicalSlug": "anthropic/claude-opus-5-20260723", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-opus-5-20260723", "model_id": "anthropic/claude-opus-5", "model_name": "Claude Opus 5", "context_length": 1000000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/us-east-1", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-5@google", "name": "Claude Opus 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-5", "canonicalSlug": "anthropic/claude-opus-5-20260723", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-opus-5-20260723", "model_id": "anthropic/claude-opus-5", "model_name": "Claude Opus 5", "context_length": 1000000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/europe", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-5@google", "name": "Claude Opus 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-5", "canonicalSlug": "anthropic/claude-opus-5-20260723", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-opus-5-20260723", "model_id": "anthropic/claude-opus-5", "model_name": "Claude Opus 5", "context_length": 1000000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-5:batch@anthropic", "name": "Claude Opus 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [ { "amount": 3.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-5:batch", "canonicalSlug": "anthropic/claude-opus-5-20260723", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-opus-5-20260723:batch", "model_id": "anthropic/claude-opus-5:batch", "model_name": "Claude Opus 5", "context_length": 1000000, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "input_cache_write_1h": "0.000005", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "microsoft/mai-image-2.5-pro@azure", "name": "Microsoft: MAI-Image-2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "microsoft/mai-image-2.5-pro", "canonicalSlug": "microsoft/mai-image-2.5-pro-20260723", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 4096, "maxCompletionTokens": 1024, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "max_completion_tokens" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Microsoft's MAI-Image-2.5 is a high-quality image generation model available via Azure AI Foundry. It produces photorealistic and artistic images from text prompts with support for various aspect ratios.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | microsoft/mai-image-2.5-pro-20260723", "model_id": "microsoft/mai-image-2.5-pro", "model_name": "Microsoft: MAI-Image-2.5 Pro", "context_length": 4096, "pricing": { "prompt": "0.000005", "completion": "0", "image_token": "0.000108", "image_output": "0.000108", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 1024, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "max_completion_tokens" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "microsoft/mai-voice-2-flash@azure", "name": "Microsoft: MAI-Voice-2-Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "microsoft/mai-voice-2-flash", "canonicalSlug": "microsoft/mai-voice-2-flash-20260723", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "max_completion_tokens" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "MAI-Voice-2-Flash is a low-latency text-to-speech model from Microsoft for voice agents, assistants, call centers, accessibility, narration, and other interactive applications. It generates expressive 24 kHz mono speech across 15 languages...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | microsoft/mai-voice-2-flash-20260723", "model_id": "microsoft/mai-voice-2-flash", "model_name": "Microsoft: MAI-Voice-2-Flash", "context_length": 0, "pricing": { "prompt": "0.000015", "completion": "0", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "max_completion_tokens" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "inclusionai/ling-3.0-flash@novita", "name": "Ling-3.0-flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.020999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000021" } ], "output": [ { "amount": 0.063, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000063" } ], "cacheRead": [ { "amount": 0.004200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000042" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "inclusionai/ling-3.0-flash", "canonicalSlug": "inclusionai/ling-3.0-flash-20260723", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | inclusionai/ling-3.0-flash-20260723", "model_id": "inclusionai/ling-3.0-flash", "model_name": "Ling-3.0-flash", "context_length": 262144, "pricing": { "prompt": "0.000000021", "completion": "0.000000063", "input_cache_read": "0.0000000042", "discount": 0.65 }, "provider_name": "Novita", "tag": "novita", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99953517450712, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "inclusionai/ling-3.0-flash@deepinfra", "name": "Ling-3.0-flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "cacheRead": [ { "amount": 0.012, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "inclusionai/ling-3.0-flash", "canonicalSlug": "inclusionai/ling-3.0-flash-20260723", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | inclusionai/ling-3.0-flash-20260723", "model_id": "inclusionai/ling-3.0-flash", "model_name": "Ling-3.0-flash", "context_length": 131072, "pricing": { "prompt": "0.00000006", "completion": "0.00000018", "input_cache_read": "0.000000012", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 98.69006866256406, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-audio-3.0-tts-flash@alibaba", "name": "Qwen: Qwen-Audio-3.0-TTS Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-audio-3.0-tts-flash", "canonicalSlug": "qwen/qwen-audio-3.0-tts-flash-20260723", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "Qwen-Audio-3.0-TTS Flash is Alibaba's fast, cost-efficient text-to-speech model, generating spoken audio from text via the DashScope Speech Synthesizer API.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen-audio-3.0-tts-flash-20260723", "model_id": "qwen/qwen-audio-3.0-tts-flash", "model_name": "Qwen: Qwen-Audio-3.0-TTS Flash", "context_length": 0, "pricing": { "prompt": "0.000015", "completion": "0", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-audio-3.0-tts-plus@alibaba", "name": "Qwen: Qwen-Audio-3.0-TTS Plus", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-audio-3.0-tts-plus", "canonicalSlug": "qwen/qwen-audio-3.0-tts-plus-20260723", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "Qwen-Audio-3.0-TTS Plus is Alibaba's higher-quality text-to-speech model, generating spoken audio from text via the DashScope Speech Synthesizer API.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen-audio-3.0-tts-plus-20260723", "model_id": "qwen/qwen-audio-3.0-tts-plus", "model_name": "Qwen: Qwen-Audio-3.0-TTS Plus", "context_length": 0, "pricing": { "prompt": "0.00002", "completion": "0", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-stt-1.0@xai", "name": "SpaceXAI: Grok STT 1.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 100000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.1" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-stt-1.0", "canonicalSlug": "x-ai/grok-stt-20260723", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok STT is SpaceXAI's speech-to-text model, available via the REST /v1/stt endpoint. It supports transcription with word-level timestamps, optional speaker diarization, and multichannel audio.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-stt-20260723", "model_id": "x-ai/grok-stt-1.0", "model_name": "SpaceXAI: Grok STT 1.0", "context_length": 0, "pricing": { "prompt": "0.1", "completion": "0", "discount": 0 }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "poolside/laguna-s-2.1@poolside", "name": "Poolside: Laguna S 2.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "cacheRead": [ { "amount": 0.009, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000009" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "poolside/laguna-s-2.1", "canonicalSlug": "poolside/laguna-s-2.1-20260720", "servingProvider": "Poolside", "servingProviderSlug": "poolside", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "temperature", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Laguna S 2.1 is the latest coding agent model from [Poolside](). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Poolside | poolside/laguna-s-2.1-20260720", "model_id": "poolside/laguna-s-2.1", "model_name": "Poolside: Laguna S 2.1", "context_length": 1048576, "pricing": { "prompt": "0.00000009", "completion": "0.00000018", "input_cache_read": "0.000000009", "discount": 0.1 }, "provider_name": "Poolside", "tag": "poolside/fp4", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "temperature", "max_tokens" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "poolside/laguna-s-2.1:free@poolside", "name": "Poolside: Laguna S 2.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "poolside/laguna-s-2.1:free", "canonicalSlug": "poolside/laguna-s-2.1-20260720", "servingProvider": "Poolside", "servingProviderSlug": "poolside", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Laguna S 2.1 is the latest coding agent model from [Poolside](). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Poolside | poolside/laguna-s-2.1-20260720:free", "model_id": "poolside/laguna-s-2.1:free", "model_name": "Poolside: Laguna S 2.1", "context_length": 262144, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Poolside", "tag": "poolside/fp4", "quantization": "fp4", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.94254482240457, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.6-flash@google", "name": "Google: Gemini 3.6 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 7.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000075" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.6-flash", "canonicalSlug": "google/gemini-3.6-flash-20260721", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...", "uptimeLast30m": 97.72962528465945, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.6-flash-20260721", "model_id": "google/gemini-3.6-flash", "model_name": "Google: Gemini 3.6 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000075", "completion": "0.00000375", "image": "0.00000075", "audio": "0.00000075", "input_audio_cache": "0.000000075", "web_search": "0.014", "internal_reasoning": "0.00000375", "input_cache_read": "0.000000075", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.72962528465945, "uptime_last_5m": 96.4471403812825, "uptime_last_1d": 98.78905797080957, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.6-flash@google", "name": "Google: Gemini 3.6 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "output": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "cacheWrite": [ { "amount": 0.0208333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000208333333333333" } ], "other": [ { "amount": 3.75e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000375" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.6-flash", "canonicalSlug": "google/gemini-3.6-flash-20260721", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...", "uptimeLast30m": 97.72962528465945, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.6-flash-20260721", "model_id": "google/gemini-3.6-flash", "model_name": "Google: Gemini 3.6 Flash", "context_length": 1048576, "pricing": { "prompt": "0.000000375", "completion": "0.000001875", "image": "0.000000375", "audio": "0.000000375", "input_audio_cache": "0.0000000375", "web_search": "0.014", "internal_reasoning": "0.000001875", "input_cache_read": "0.0000000375", "input_cache_write": "0.0000000208333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.72962528465945, "uptime_last_5m": 96.4471403812825, "uptime_last_1d": 98.78905797080957, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.6-flash@google", "name": "Google: Gemini 3.6 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 6.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000675" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000135" } ], "cacheWrite": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "other": [ { "amount": 0.00000135, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000135" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 6.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000675" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.6-flash", "canonicalSlug": "google/gemini-3.6-flash-20260721", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...", "uptimeLast30m": 97.72962528465945, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.6-flash-20260721", "model_id": "google/gemini-3.6-flash", "model_name": "Google: Gemini 3.6 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000135", "completion": "0.00000675", "image": "0.00000135", "audio": "0.00000135", "input_audio_cache": "0.000000135", "web_search": "0.014", "internal_reasoning": "0.00000675", "input_cache_read": "0.000000135", "input_cache_write": "0.000000075", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.72962528465945, "uptime_last_5m": 96.4471403812825, "uptime_last_1d": 98.78905797080957, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.6-flash@google-ai-studio", "name": "Google: Gemini 3.6 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 7.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000075" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.6-flash", "canonicalSlug": "google/gemini-3.6-flash-20260721", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...", "uptimeLast30m": 97.52149330028658, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.6-flash-20260721", "model_id": "google/gemini-3.6-flash", "model_name": "Google: Gemini 3.6 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000075", "completion": "0.00000375", "image": "0.00000075", "audio": "0.00000075", "input_audio_cache": "0.000000075", "web_search": "0.014", "internal_reasoning": "0.00000375", "input_cache_read": "0.000000075", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.52149330028658, "uptime_last_5m": 98.54479308776718, "uptime_last_1d": 99.1265536896769, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.6-flash@google-ai-studio", "name": "Google: Gemini 3.6 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "output": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "cacheWrite": [ { "amount": 0.0208333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000208333333333333" } ], "other": [ { "amount": 3.75e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000375" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.6-flash", "canonicalSlug": "google/gemini-3.6-flash-20260721", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...", "uptimeLast30m": 97.52149330028658, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.6-flash-20260721", "model_id": "google/gemini-3.6-flash", "model_name": "Google: Gemini 3.6 Flash", "context_length": 1048576, "pricing": { "prompt": "0.000000375", "completion": "0.000001875", "image": "0.000000375", "audio": "0.000000375", "input_audio_cache": "0.0000000375", "web_search": "0.014", "internal_reasoning": "0.000001875", "input_cache_read": "0.0000000375", "input_cache_write": "0.0000000208333333333333", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.52149330028658, "uptime_last_5m": 98.54479308776718, "uptime_last_1d": 99.1265536896769, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.6-flash@google-ai-studio", "name": "Google: Gemini 3.6 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 6.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000675" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000135" } ], "cacheWrite": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "other": [ { "amount": 0.00000135, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000135" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 6.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000675" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.6-flash", "canonicalSlug": "google/gemini-3.6-flash-20260721", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...", "uptimeLast30m": 97.52149330028658, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.6-flash-20260721", "model_id": "google/gemini-3.6-flash", "model_name": "Google: Gemini 3.6 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000135", "completion": "0.00000675", "image": "0.00000135", "audio": "0.00000135", "input_audio_cache": "0.000000135", "web_search": "0.014", "internal_reasoning": "0.00000675", "input_cache_read": "0.000000135", "input_cache_write": "0.000000075", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.52149330028658, "uptime_last_5m": 98.54479308776718, "uptime_last_1d": 99.1265536896769, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.6-flash@google", "name": "Google: Gemini 3.6 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.8250000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000825" } ], "output": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" } ], "cacheRead": [ { "amount": 0.0825, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000825" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 8.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000825" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.6-flash", "canonicalSlug": "google/gemini-3.6-flash-20260721", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.6-flash-20260721", "model_id": "google/gemini-3.6-flash", "model_name": "Google: Gemini 3.6 Flash", "context_length": 1048576, "pricing": { "prompt": "0.000000825", "completion": "0.000004125", "image": "0.000000825", "audio": "0.000000825", "input_audio_cache": "0.0000000825", "web_search": "0.014", "internal_reasoning": "0.000004125", "input_cache_read": "0.0000000825", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.6-flash:batch@google", "name": "Google: Gemini 3.6 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "output": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 3.75e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000375" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.6-flash:batch", "canonicalSlug": "google/gemini-3.6-flash-20260721", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.6-flash-20260721:batch", "model_id": "google/gemini-3.6-flash:batch", "model_name": "Google: Gemini 3.6 Flash", "context_length": 1048576, "pricing": { "prompt": "0.000000375", "completion": "0.000001875", "image": "0.000000375", "audio": "0.000000375", "input_audio_cache": "0.0000000375", "web_search": "0.014", "internal_reasoning": "0.000001875", "input_cache_read": "0.0000000375", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash-lite@google-ai-studio", "name": "Google: Gemini 3.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 3e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000003" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash-lite", "canonicalSlug": "google/gemini-3.5-flash-lite-20260721", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.", "uptimeLast30m": 99.93816600366881, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.5-flash-lite-20260721", "model_id": "google/gemini-3.5-flash-lite", "model_name": "Google: Gemini 3.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "image": "0.0000003", "audio": "0.0000003", "input_audio_cache": "0.00000003", "web_search": "0.014", "internal_reasoning": "0.0000025", "input_cache_read": "0.00000003", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.93816600366881, "uptime_last_5m": 99.98298451590948, "uptime_last_1d": 99.91162805678351, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash-lite@google-ai-studio", "name": "Google: Gemini 3.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000015" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 1.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000015" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash-lite", "canonicalSlug": "google/gemini-3.5-flash-lite-20260721", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.", "uptimeLast30m": 99.93816600366881, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.5-flash-lite-20260721", "model_id": "google/gemini-3.5-flash-lite", "model_name": "Google: Gemini 3.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000015", "completion": "0.00000125", "image": "0.00000015", "audio": "0.00000015", "input_audio_cache": "0.000000015", "web_search": "0.014", "internal_reasoning": "0.00000125", "input_cache_read": "0.000000015", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.93816600366881, "uptime_last_5m": 99.98298451590948, "uptime_last_1d": 99.91162805678351, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash-lite@google-ai-studio", "name": "Google: Gemini 3.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000054" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.054, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000054" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 5.4e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000054" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash-lite", "canonicalSlug": "google/gemini-3.5-flash-lite-20260721", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.", "uptimeLast30m": 99.93816600366881, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.5-flash-lite-20260721", "model_id": "google/gemini-3.5-flash-lite", "model_name": "Google: Gemini 3.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000054", "completion": "0.0000045", "image": "0.00000054", "audio": "0.00000054", "input_audio_cache": "0.000000054", "web_search": "0.014", "internal_reasoning": "0.0000045", "input_cache_read": "0.000000054", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.93816600366881, "uptime_last_5m": 99.98298451590948, "uptime_last_1d": 99.91162805678351, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash-lite@google", "name": "Google: Gemini 3.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 3e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000003" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash-lite", "canonicalSlug": "google/gemini-3.5-flash-lite-20260721", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.", "uptimeLast30m": 99.96205413609917, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.5-flash-lite-20260721", "model_id": "google/gemini-3.5-flash-lite", "model_name": "Google: Gemini 3.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "image": "0.0000003", "audio": "0.0000003", "input_audio_cache": "0.00000003", "web_search": "0.014", "internal_reasoning": "0.0000025", "input_cache_read": "0.00000003", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.96205413609917, "uptime_last_5m": 99.96589358799454, "uptime_last_1d": 99.92823514217953, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash-lite@google", "name": "Google: Gemini 3.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000015" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 1.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000015" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash-lite", "canonicalSlug": "google/gemini-3.5-flash-lite-20260721", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.", "uptimeLast30m": 99.96205413609917, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.5-flash-lite-20260721", "model_id": "google/gemini-3.5-flash-lite", "model_name": "Google: Gemini 3.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000015", "completion": "0.00000125", "image": "0.00000015", "audio": "0.00000015", "input_audio_cache": "0.000000015", "web_search": "0.014", "internal_reasoning": "0.00000125", "input_cache_read": "0.000000015", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.96205413609917, "uptime_last_5m": 99.96589358799454, "uptime_last_1d": 99.92823514217953, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash-lite@google", "name": "Google: Gemini 3.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000054" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.054, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000054" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 5.4e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000054" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash-lite", "canonicalSlug": "google/gemini-3.5-flash-lite-20260721", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.", "uptimeLast30m": 99.96205413609917, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.5-flash-lite-20260721", "model_id": "google/gemini-3.5-flash-lite", "model_name": "Google: Gemini 3.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000054", "completion": "0.0000045", "image": "0.00000054", "audio": "0.00000054", "input_audio_cache": "0.000000054", "web_search": "0.014", "internal_reasoning": "0.0000045", "input_cache_read": "0.000000054", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.96205413609917, "uptime_last_5m": 99.96589358799454, "uptime_last_1d": 99.92823514217953, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash-lite@google", "name": "Google: Gemini 3.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000033" } ], "output": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "cacheRead": [ { "amount": 0.032999999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000033" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 3.3e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000033" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash-lite", "canonicalSlug": "google/gemini-3.5-flash-lite-20260721", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.5-flash-lite-20260721", "model_id": "google/gemini-3.5-flash-lite", "model_name": "Google: Gemini 3.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000033", "completion": "0.00000275", "image": "0.00000033", "audio": "0.00000033", "input_audio_cache": "0.000000033", "web_search": "0.014", "internal_reasoning": "0.00000275", "input_cache_read": "0.000000033", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash-lite:batch@google", "name": "Google: Gemini 3.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000015" } ], "cacheWrite": [], "other": [ { "amount": 1.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000015" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash-lite:batch", "canonicalSlug": "google/gemini-3.5-flash-lite-20260721", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.5-flash-lite-20260721:batch", "model_id": "google/gemini-3.5-flash-lite:batch", "model_name": "Google: Gemini 3.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000015", "completion": "0.00000125", "image": "0.00000015", "audio": "0.00000015", "input_audio_cache": "0.000000015", "web_search": "0.014", "internal_reasoning": "0.00000125", "input_cache_read": "0.000000015", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "krea/krea-2-large@krea", "name": "Krea: Krea 2 Large", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "krea/krea-2-large", "canonicalSlug": "krea/krea-2-large-20260720", "servingProvider": "Krea", "servingProviderSlug": "krea", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Media", "instruct_type": null }, "description": "Krea 2 Large is Krea's high-capability image generation model, more than twice the size of Krea 2 Medium. Its lighter post-training gives images a rawer, more textured, and flexible character,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Krea | krea/krea-2-large-20260720", "model_id": "krea/krea-2-large", "model_name": "Krea: Krea 2 Large", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.0000143712574850299", "image_output": "0.0000143712574850299", "discount": 0 }, "provider_name": "Krea", "tag": "krea", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "krea/krea-2-medium@krea", "name": "Krea: Krea 2 Medium", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "krea/krea-2-medium", "canonicalSlug": "krea/krea-2-medium-20260720", "servingProvider": "Krea", "servingProviderSlug": "krea", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Media", "instruct_type": null }, "description": "Krea 2 Medium is Krea's balanced, cost-efficient image generation model and a practical starting point for a broad range of use cases. Its extensive post-training supports stable, consistent generations, with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Krea | krea/krea-2-medium-20260720", "model_id": "krea/krea-2-medium", "model_name": "Krea: Krea 2 Medium", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.00000718562874251497", "image_output": "0.00000718562874251497", "discount": 0 }, "provider_name": "Krea", "tag": "krea", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "krea/krea-2-medium-turbo@krea", "name": "Krea: Krea 2 Medium Turbo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "krea/krea-2-medium-turbo", "canonicalSlug": "krea/krea-2-medium-turbo-20260720", "servingProvider": "Krea", "servingProviderSlug": "krea", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Media", "instruct_type": null }, "description": "Krea 2 Medium Turbo is a distilled, speed-focused variant of Krea 2 Medium from Krea. It is designed for rapid iteration and graphic design exploration where fast generation is the...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Krea | krea/krea-2-medium-turbo-20260720", "model_id": "krea/krea-2-medium-turbo", "model_name": "Krea: Krea 2 Medium Turbo", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.00000359281437125749", "image_output": "0.00000359281437125749", "discount": 0 }, "provider_name": "Krea", "tag": "krea", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meituan/longcat-2.0@atlascloud", "name": "Meituan: LongCat 2.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.006, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meituan/longcat-2.0", "canonicalSlug": "meituan/longcat-2.0-20260720", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 1048756, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | meituan/longcat-2.0-20260720", "model_id": "meituan/longcat-2.0", "model_name": "Meituan: LongCat 2.0", "context_length": 1048756, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.000000006", "discount": 0.6 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-imagine-video-1.5@xai", "name": "SpaceXAI: Grok Imagine Video 1.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-imagine-video-1.5", "canonicalSlug": "x-ai/grok-imagine-video-1.5-20260719", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok Imagine Video 1.5 is a video generation model from SpaceXAI. It creates videos from text prompts, with an optional starting image to guide the scene. It can direct subject...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-imagine-video-1.5-20260719", "model_id": "x-ai/grok-imagine-video-1.5", "model_name": "SpaceXAI: Grok Imagine Video 1.5", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thinkingmachines/inkling@deepinfra", "name": "Thinking Machines: Inkling", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000405" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thinkingmachines/inkling", "canonicalSlug": "thinkingmachines/inkling-20260715", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 524288, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+audio->text", "input_modalities": [ "text", "image", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...", "uptimeLast30m": 99.4731296101159, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | thinkingmachines/inkling-20260715", "model_id": "thinkingmachines/inkling", "model_name": "Thinking Machines: Inkling", "context_length": 524288, "pricing": { "prompt": "0.00000095", "completion": "0.00000405", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.4731296101159, "uptime_last_5m": 100, "uptime_last_1d": 99.65298459147999, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thinkingmachines/inkling@baseten", "name": "Thinking Machines: Inkling", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 4.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000405" } ], "cacheRead": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000017" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thinkingmachines/inkling", "canonicalSlug": "thinkingmachines/inkling-20260715", "servingProvider": "BaseTen", "servingProviderSlug": "baseten", "contextLength": 1048576, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+audio->text", "input_modalities": [ "text", "image", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...", "uptimeLast30m": 99.85228951255539, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "BaseTen | thinkingmachines/inkling-20260715", "model_id": "thinkingmachines/inkling", "model_name": "Thinking Machines: Inkling", "context_length": 1048576, "pricing": { "prompt": "0.000001", "completion": "0.00000405", "input_cache_read": "0.00000017", "discount": 0 }, "provider_name": "BaseTen", "tag": "baseten/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.85228951255539, "uptime_last_5m": 100, "uptime_last_1d": 99.75997256829352, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thinkingmachines/inkling@together", "name": "Thinking Machines: Inkling", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 4.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000405" } ], "cacheRead": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000017" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thinkingmachines/inkling", "canonicalSlug": "thinkingmachines/inkling-20260715", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 524288, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+audio->text", "input_modalities": [ "text", "image", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...", "uptimeLast30m": 97.35571878279119, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | thinkingmachines/inkling-20260715", "model_id": "thinkingmachines/inkling", "model_name": "Thinking Machines: Inkling", "context_length": 524288, "pricing": { "prompt": "0.000001", "completion": "0.00000405", "input_cache_read": "0.00000017", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.35571878279119, "uptime_last_5m": 98.54014598540147, "uptime_last_1d": 97.54701863079728, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thinkingmachines/inkling:batch@together", "name": "Thinking Machines: Inkling", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 4.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000405" } ], "cacheRead": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000017" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thinkingmachines/inkling:batch", "canonicalSlug": "thinkingmachines/inkling-20260715", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 524288, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+audio->text", "input_modalities": [ "text", "image", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | thinkingmachines/inkling-20260715:batch", "model_id": "thinkingmachines/inkling:batch", "model_name": "Thinking Machines: Inkling", "context_length": 524288, "pricing": { "prompt": "0.000001", "completion": "0.00000405", "input_cache_read": "0.00000017", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openrouter/auto-beta", "name": "Auto Router (Beta)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": -1000000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "-1" } ], "output": [ { "amount": -1000000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "-1" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "canonicalSlug": "openrouter/auto-beta", "contextLength": 2000000, "architecture": { "modality": "text+image+file+audio+video->text+image", "input_modalities": [ "text", "image", "audio", "file", "video" ], "output_modalities": [ "text", "image" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "prediction", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_a", "top_k", "top_logprobs", "top_p", "web_search_options" ], "description": "Auto Router (Beta) is a task-aware router from OpenRouter. It classifies each request, then routes it the [most popular model](/rankings#task-spend) for that task based on aggregate spend, filtered by your...", "endpointCount": 0 } }, { "id": "deepgram/aura-2@deepgram", "name": "Deepgram: Aura-2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00003" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepgram/aura-2", "canonicalSlug": "deepgram/aura-2-20260716", "servingProvider": "Deepgram", "servingProviderSlug": "deepgram", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "Aura-2 is a multilingual text-to-speech model from Deepgram. It supports Deepgram’s canonical Aura-2 voice catalog for speech synthesis across multiple languages.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Deepgram | deepgram/aura-2-20260716", "model_id": "deepgram/aura-2", "model_name": "Deepgram: Aura-2", "context_length": 0, "pricing": { "prompt": "0.00003", "completion": "0", "discount": 0 }, "provider_name": "Deepgram", "tag": "deepgram", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@sail-research", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000026" } ], "output": [ { "amount": 13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000013" } ], "cacheRead": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000029" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "Sail Research", "servingProviderSlug": "sail-research", "contextLength": 974842, "maxCompletionTokens": 974842, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "tools", "tool_choice", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 98.58254585881045, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Sail Research | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 974842, "pricing": { "prompt": "0.0000026", "completion": "0.000013", "input_cache_read": "0.00000029", "discount": 0 }, "provider_name": "Sail Research", "tag": "sail-research/fp4", "quantization": "fp4", "max_completion_tokens": 974842, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "tools", "tool_choice", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.58254585881045, "uptime_last_5m": 99.13793103448276, "uptime_last_1d": 99.01455956134137, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@morph", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000028" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000029" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "Morph", "servingProviderSlug": "morph", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 95.52715654952077, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Morph | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1048576, "pricing": { "prompt": "0.0000028", "completion": "0.000014", "input_cache_read": "0.00000029", "discount": 0 }, "provider_name": "Morph", "tag": "morph/fp4", "quantization": "fp4", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 95.52715654952077, "uptime_last_5m": 92.37288135593221, "uptime_last_1d": 94.98753850678803, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@digitalocean", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.8499999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000285" } ], "output": [ { "amount": 14.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001425" } ], "cacheRead": [ { "amount": 0.28500000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000285" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 99.52651515151516, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1048576, "pricing": { "prompt": "0.00000285", "completion": "0.00001425", "input_cache_read": "0.000000285", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.52651515151516, "uptime_last_5m": 99.47712418300654, "uptime_last_1d": 97.62644842714093, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@deepinfra", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.8499999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000285" } ], "output": [ { "amount": 14.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001425" } ], "cacheRead": [ { "amount": 0.28500000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000285" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 1048576, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 96.18717504332756, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1048576, "pricing": { "prompt": "0.00000285", "completion": "0.00001425", "input_cache_read": "0.000000285", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 96.18717504332756, "uptime_last_5m": null, "uptime_last_1d": 97.89877097925202, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@fireworks", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "Fireworks", "servingProviderSlug": "fireworks", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 99.12854030501089, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fireworks | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1048576, "pricing": { "prompt": "0.000003", "completion": "0.000015", "input_cache_read": "0.0000003", "discount": 0 }, "provider_name": "Fireworks", "tag": "fireworks", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.12854030501089, "uptime_last_5m": 100, "uptime_last_1d": 98.32203988567392, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@chutes", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "Chutes", "servingProviderSlug": "chutes", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "mxfp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Chutes | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1048576, "pricing": { "prompt": "0.000003", "completion": "0.000015", "input_cache_read": "0.0000003", "discount": 0 }, "provider_name": "Chutes", "tag": "chutes/mxfp4", "quantization": "mxfp4", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 95.40595458962807, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@together", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 1000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 99.14984059511158, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "input_cache_read": "0.0000003", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.14984059511158, "uptime_last_5m": 99.62779156327544, "uptime_last_1d": 98.79771179210316, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@moonshot-ai", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "Moonshot AI", "servingProviderSlug": "moonshot-ai", "contextLength": 1048576, "quantization": "mxfp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "frequency_penalty", "presence_penalty", "structured_outputs", "response_format", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 99.47729278392505, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Moonshot AI | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1048576, "pricing": { "prompt": "0.000003", "completion": "0.000015", "input_cache_read": "0.0000003", "discount": 0 }, "provider_name": "Moonshot AI", "tag": "moonshotai/mxfp4", "quantization": "mxfp4", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "frequency_penalty", "presence_penalty", "structured_outputs", "response_format", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.47729278392505, "uptime_last_5m": 99.57023905452593, "uptime_last_1d": 99.76835977783477, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@wafer", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "Wafer", "servingProviderSlug": "wafer", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 99.079754601227, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Wafer | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1048576, "pricing": { "prompt": "0.000003", "completion": "0.000015", "input_cache_read": "0.0000003", "discount": 0 }, "provider_name": "Wafer", "tag": "wafer", "quantization": "unknown", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.079754601227, "uptime_last_5m": 100, "uptime_last_1d": 95.56412729026037, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@modal", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "Modal", "servingProviderSlug": "modal", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "mxfp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "stop", "max_tokens", "response_format", "structured_outputs", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 99.6809386578839, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Modal | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1048576, "pricing": { "prompt": "0.000003", "completion": "0.000015", "input_cache_read": "0.0000003", "discount": 0 }, "provider_name": "Modal", "tag": "modal/mxfp4", "quantization": "mxfp4", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "stop", "max_tokens", "response_format", "structured_outputs", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.6809386578839, "uptime_last_5m": 99.91103202846975, "uptime_last_1d": 99.062993222542, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@baseten", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "BaseTen", "servingProviderSlug": "baseten", "contextLength": 1048576, "maxCompletionTokens": 262144, "quantization": "fp8", "status": -5, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 79.62962962962963, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "BaseTen | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1048576, "pricing": { "prompt": "0.000003", "completion": "0.000015", "input_cache_read": "0.0000003", "discount": 0 }, "provider_name": "BaseTen", "tag": "baseten/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "status": -5, "uptime_last_30m": 79.62962962962963, "uptime_last_5m": null, "uptime_last_1d": 94.52091053467443, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@phala", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 99.2467043314501, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1048576, "pricing": { "prompt": "0.000003", "completion": "0.000015", "input_cache_read": "0.0000015", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.2467043314501, "uptime_last_5m": 99.06832298136646, "uptime_last_1d": 97.2504457468499, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k3@morph", "name": "MoonshotAI: Kimi K3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k3", "canonicalSlug": "moonshotai/kimi-k3-20260715", "servingProvider": "Morph", "servingProviderSlug": "morph", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "tools", "logprobs", "top_logprobs", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Morph | moonshotai/kimi-k3-20260715", "model_id": "moonshotai/kimi-k3", "model_name": "MoonshotAI: Kimi K3", "context_length": 1048576, "pricing": { "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "discount": 0 }, "provider_name": "Morph", "tag": "morph/fast", "quantization": "fp4", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "tools", "logprobs", "top_logprobs", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 94.69560842377632, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta/muse-spark-1.1@meta", "name": "Meta: Muse Spark 1.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 4.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000425" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [ { "amount": 0.0025, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.0025" } ] } ], "metadata": { "source": "openrouter", "modelId": "meta/muse-spark-1.1", "canonicalSlug": "meta/muse-spark-1.1-20260709", "servingProvider": "Meta", "servingProviderSlug": "meta", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "repetition_penalty", "top_k", "temperature", "top_p", "tools", "tool_choice", "structured_outputs", "response_format", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Meta | meta/muse-spark-1.1-20260709", "model_id": "meta/muse-spark-1.1", "model_name": "Meta: Muse Spark 1.1", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00000425", "web_search": "0.0025", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "Meta", "tag": "meta", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "repetition_penalty", "top_k", "temperature", "top_p", "tools", "tool_choice", "structured_outputs", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-embed-1b:free@nvidia", "name": "NVIDIA: Nemotron 3 Embed 1B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-embed-1b:free", "canonicalSlug": "nvidia/nemotron-3-embed-1b-20260716", "servingProvider": "Nvidia", "servingProviderSlug": "nvidia", "contextLength": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "temperature", "max_tokens", "seed", "top_p" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Embed 1B is an open text embedding model from NVIDIA, optimized for high-throughput, low-latency retrieval. It is suited for enterprise search, RAG, code retrieval, and agentic retrieval...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nvidia | nvidia/nemotron-3-embed-1b-20260716:free", "model_id": "nvidia/nemotron-3-embed-1b:free", "model_name": "NVIDIA: Nemotron 3 Embed 1B", "context_length": 32768, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Nvidia", "tag": "nvidia", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "temperature", "max_tokens", "seed", "top_p" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/speech-2.8-hd@minimax", "name": "MiniMax: Speech 2.8 HD", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 100, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/speech-2.8-hd", "canonicalSlug": "minimax/speech-2.8-hd-20260716", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax Speech 2.8 HD is a text-to-speech model from MiniMax. It is suited for applications that generate spoken audio from text and accepts arbitrary MiniMax voice IDs.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/speech-2.8-hd-20260716", "model_id": "minimax/speech-2.8-hd", "model_name": "MiniMax: Speech 2.8 HD", "context_length": 0, "pricing": { "prompt": "0.0001", "completion": "0", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/speech-2.8-turbo@minimax", "name": "MiniMax: Speech 2.8 Turbo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00006" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/speech-2.8-turbo", "canonicalSlug": "minimax/speech-2.8-turbo-20260716", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax Speech 2.8 Turbo is a text-to-speech model from MiniMax. It is suited for applications that generate spoken audio from text and accepts arbitrary MiniMax voice IDs.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/speech-2.8-turbo-20260716", "model_id": "minimax/speech-2.8-turbo", "model_name": "MiniMax: Speech 2.8 Turbo", "context_length": 0, "pricing": { "prompt": "0.00006", "completion": "0", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepgram/nova-3@deepgram", "name": "Deepgram: Nova-3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 4300, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0043" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepgram/nova-3", "canonicalSlug": "deepgram/nova-3-20260714", "servingProvider": "Deepgram", "servingProviderSlug": "deepgram", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Other", "instruct_type": null }, "description": "Deepgram Nova-3 general-purpose speech-to-text model with monolingual and multilingual transcription support.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Deepgram | deepgram/nova-3-20260714", "model_id": "deepgram/nova-3", "model_name": "Deepgram: Nova-3", "context_length": 0, "pricing": { "prompt": "0.0043", "completion": "0", "discount": 0 }, "provider_name": "Deepgram", "tag": "deepgram", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "kwaipilot/kat-coder-air-v2.5@streamlake", "name": "Kwaipilot: KAT-Coder-Air V2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "kwaipilot/kat-coder-air-v2.5", "canonicalSlug": "kwaipilot/kat-coder-air-v2.5-20260710", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 256000, "maxCompletionTokens": 80000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "presence_penalty", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "KAT-Coder-Air V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | kwaipilot/kat-coder-air-v2.5-20260710", "model_id": "kwaipilot/kat-coder-air-v2.5", "model_name": "Kwaipilot: KAT-Coder-Air V2.5", "context_length": 256000, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 80000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "presence_penalty", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "kwaipilot/kat-coder-pro-v2.5@streamlake", "name": "Kwaipilot: KAT-Coder-Pro V2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000074" } ], "output": [ { "amount": 2.96, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000296" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "kwaipilot/kat-coder-pro-v2.5", "canonicalSlug": "kwaipilot/kat-coder-pro-v2.5-20260710", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 256000, "maxCompletionTokens": 80000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "presence_penalty", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | kwaipilot/kat-coder-pro-v2.5-20260710", "model_id": "kwaipilot/kat-coder-pro-v2.5", "model_name": "Kwaipilot: KAT-Coder-Pro V2.5", "context_length": 256000, "pricing": { "prompt": "0.00000074", "completion": "0.00000296", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 80000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "presence_penalty", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna-pro@openai", "name": "OpenAI: GPT-5.6 Luna Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna-pro", "canonicalSlug": "openai/gpt-5.6-luna-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": 99.96451164074514, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-pro-20260709", "model_id": "openai/gpt-5.6-luna-pro", "model_name": "OpenAI: GPT-5.6 Luna Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000002", "completion": "0.0000012", "web_search": "0.01", "input_cache_read": "0.00000002", "input_cache_write": "0.00000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000004", "completion": "0.0000018", "input_cache_read": "0.00000004", "input_cache_write": "0.0000005" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.96451164074514, "uptime_last_5m": 99.97118027844284, "uptime_last_1d": 99.97586907065758, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna-pro@openai", "name": "OpenAI: GPT-5.6 Luna Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna-pro", "canonicalSlug": "openai/gpt-5.6-luna-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": 99.96451164074514, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-pro-20260709", "model_id": "openai/gpt-5.6-luna-pro", "model_name": "OpenAI: GPT-5.6 Luna Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000001", "completion": "0.0000006", "web_search": "0.01", "input_cache_read": "0.00000001", "input_cache_write": "0.000000125", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000002", "completion": "0.0000009", "input_cache_read": "0.00000002", "input_cache_write": "0.00000025" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.96451164074514, "uptime_last_5m": 99.97118027844284, "uptime_last_1d": 99.97586907065758, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna-pro@openai", "name": "OpenAI: GPT-5.6 Luna Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna-pro", "canonicalSlug": "openai/gpt-5.6-luna-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": 99.96451164074514, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-pro-20260709", "model_id": "openai/gpt-5.6-luna-pro", "model_name": "OpenAI: GPT-5.6 Luna Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000004", "completion": "0.0000024", "web_search": "0.01", "input_cache_read": "0.00000004", "input_cache_write": "0.0000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.96451164074514, "uptime_last_5m": 99.97118027844284, "uptime_last_1d": 99.97586907065758, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna-pro@azure", "name": "OpenAI: GPT-5.6 Luna Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna-pro", "canonicalSlug": "openai/gpt-5.6-luna-pro-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-luna-pro-20260709", "model_id": "openai/gpt-5.6-luna-pro", "model_name": "OpenAI: GPT-5.6 Luna Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000002", "completion": "0.0000012", "web_search": "0.01", "input_cache_read": "0.00000002", "input_cache_write": "0.00000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000004", "completion": "0.0000018", "input_cache_read": "0.00000004", "input_cache_write": "0.0000005" } ] }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 98.68317781459841, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna-pro@azure", "name": "OpenAI: GPT-5.6 Luna Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "cacheRead": [ { "amount": 0.022, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000022" } ], "cacheWrite": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna-pro", "canonicalSlug": "openai/gpt-5.6-luna-pro-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-luna-pro-20260709", "model_id": "openai/gpt-5.6-luna-pro", "model_name": "OpenAI: GPT-5.6 Luna Pro", "context_length": 1050000, "pricing": { "prompt": "0.00000022", "completion": "0.00000132", "web_search": "0.01", "input_cache_read": "0.000000022", "input_cache_write": "0.000000275", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00000044", "completion": "0.00000198", "input_cache_read": "0.000000044", "input_cache_write": "0.00000055" } ] }, "provider_name": "Azure", "tag": "azure/eu", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna-pro:batch@openai", "name": "OpenAI: GPT-5.6 Luna Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna-pro:batch", "canonicalSlug": "openai/gpt-5.6-luna-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-pro-20260709:batch", "model_id": "openai/gpt-5.6-luna-pro:batch", "model_name": "OpenAI: GPT-5.6 Luna Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000001", "completion": "0.0000006", "web_search": "0.01", "input_cache_read": "0.00000001", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000002", "completion": "0.0000009", "input_cache_read": "0.00000002" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna-pro:batch@openai", "name": "OpenAI: GPT-5.6 Luna Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna-pro:batch", "canonicalSlug": "openai/gpt-5.6-luna-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-pro-20260709:batch", "model_id": "openai/gpt-5.6-luna-pro:batch", "model_name": "OpenAI: GPT-5.6 Luna Pro", "context_length": 1050000, "pricing": { "prompt": "0.00000005", "completion": "0.0000003", "web_search": "0.01", "input_cache_read": "0.000000005", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000001", "completion": "0.00000045", "input_cache_read": "0.00000001" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna-pro:batch@openai", "name": "OpenAI: GPT-5.6 Luna Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna-pro:batch", "canonicalSlug": "openai/gpt-5.6-luna-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-pro-20260709:batch", "model_id": "openai/gpt-5.6-luna-pro:batch", "model_name": "OpenAI: GPT-5.6 Luna Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000002", "completion": "0.0000012", "web_search": "0.01", "input_cache_read": "0.00000002", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna@openai", "name": "OpenAI: GPT-5.6 Luna", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna", "canonicalSlug": "openai/gpt-5.6-luna-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "uptimeLast30m": 99.93234526846888, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-20260709", "model_id": "openai/gpt-5.6-luna", "model_name": "OpenAI: GPT-5.6 Luna", "context_length": 1050000, "pricing": { "prompt": "0.0000002", "completion": "0.0000012", "web_search": "0.01", "input_cache_read": "0.00000002", "input_cache_write": "0.00000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000004", "completion": "0.0000018", "input_cache_read": "0.00000004", "input_cache_write": "0.0000005" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.93234526846888, "uptime_last_5m": 99.9444899318614, "uptime_last_1d": 99.7860687454102, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna@openai", "name": "OpenAI: GPT-5.6 Luna", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna", "canonicalSlug": "openai/gpt-5.6-luna-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "uptimeLast30m": 99.93234526846888, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-20260709", "model_id": "openai/gpt-5.6-luna", "model_name": "OpenAI: GPT-5.6 Luna", "context_length": 1050000, "pricing": { "prompt": "0.0000001", "completion": "0.0000006", "web_search": "0.01", "input_cache_read": "0.00000001", "input_cache_write": "0.000000125", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000002", "completion": "0.0000009", "input_cache_read": "0.00000002", "input_cache_write": "0.00000025" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.93234526846888, "uptime_last_5m": 99.9444899318614, "uptime_last_1d": 99.7860687454102, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna@openai", "name": "OpenAI: GPT-5.6 Luna", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna", "canonicalSlug": "openai/gpt-5.6-luna-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "uptimeLast30m": 99.93234526846888, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-20260709", "model_id": "openai/gpt-5.6-luna", "model_name": "OpenAI: GPT-5.6 Luna", "context_length": 1050000, "pricing": { "prompt": "0.0000004", "completion": "0.0000024", "web_search": "0.01", "input_cache_read": "0.00000004", "input_cache_write": "0.0000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.93234526846888, "uptime_last_5m": 99.9444899318614, "uptime_last_1d": 99.7860687454102, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna@azure", "name": "OpenAI: GPT-5.6 Luna", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna", "canonicalSlug": "openai/gpt-5.6-luna-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "uptimeLast30m": 99.74550081803308, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-luna-20260709", "model_id": "openai/gpt-5.6-luna", "model_name": "OpenAI: GPT-5.6 Luna", "context_length": 1050000, "pricing": { "prompt": "0.0000002", "completion": "0.0000012", "web_search": "0.01", "input_cache_read": "0.00000002", "input_cache_write": "0.00000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000004", "completion": "0.0000018", "input_cache_read": "0.00000004", "input_cache_write": "0.0000005" } ] }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.74550081803308, "uptime_last_5m": 99.72337482710927, "uptime_last_1d": 89.36232769338199, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna@azure", "name": "OpenAI: GPT-5.6 Luna", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "cacheRead": [ { "amount": 0.022, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000022" } ], "cacheWrite": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna", "canonicalSlug": "openai/gpt-5.6-luna-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "uptimeLast30m": 99.76819656930923, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-luna-20260709", "model_id": "openai/gpt-5.6-luna", "model_name": "OpenAI: GPT-5.6 Luna", "context_length": 1050000, "pricing": { "prompt": "0.00000022", "completion": "0.00000132", "web_search": "0.01", "input_cache_read": "0.000000022", "input_cache_write": "0.000000275", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00000044", "completion": "0.00000198", "input_cache_read": "0.000000044", "input_cache_write": "0.00000055" } ] }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.76819656930923, "uptime_last_5m": 99.24528301886792, "uptime_last_1d": 80.48063239348929, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna@azure", "name": "OpenAI: GPT-5.6 Luna", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "cacheRead": [ { "amount": 0.022, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000022" } ], "cacheWrite": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna", "canonicalSlug": "openai/gpt-5.6-luna-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-luna-20260709", "model_id": "openai/gpt-5.6-luna", "model_name": "OpenAI: GPT-5.6 Luna", "context_length": 1050000, "pricing": { "prompt": "0.00000022", "completion": "0.00000132", "web_search": "0.01", "input_cache_read": "0.000000022", "input_cache_write": "0.000000275", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00000044", "completion": "0.00000198", "input_cache_read": "0.000000044", "input_cache_write": "0.00000055" } ] }, "provider_name": "Azure", "tag": "azure/eu", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna@amazon-bedrock", "name": "OpenAI: GPT-5.6 Luna", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "cacheRead": [ { "amount": 0.022, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000022" } ], "cacheWrite": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna", "canonicalSlug": "openai/gpt-5.6-luna-20260709", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | openai/gpt-5.6-luna-20260709", "model_id": "openai/gpt-5.6-luna", "model_name": "OpenAI: GPT-5.6 Luna", "context_length": 1050000, "pricing": { "prompt": "0.00000022", "completion": "0.00000132", "web_search": "0.01", "input_cache_read": "0.000000022", "input_cache_write": "0.000000275", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00000044", "completion": "0.00000198", "input_cache_read": "0.000000044", "input_cache_write": "0.00000055" } ] }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/us-east-1", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9914993454496, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna:batch@openai", "name": "OpenAI: GPT-5.6 Luna", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna:batch", "canonicalSlug": "openai/gpt-5.6-luna-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-20260709:batch", "model_id": "openai/gpt-5.6-luna:batch", "model_name": "OpenAI: GPT-5.6 Luna", "context_length": 1050000, "pricing": { "prompt": "0.0000001", "completion": "0.0000006", "web_search": "0.01", "input_cache_read": "0.00000001", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000002", "completion": "0.0000009", "input_cache_read": "0.00000002" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna:batch@openai", "name": "OpenAI: GPT-5.6 Luna", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna:batch", "canonicalSlug": "openai/gpt-5.6-luna-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-20260709:batch", "model_id": "openai/gpt-5.6-luna:batch", "model_name": "OpenAI: GPT-5.6 Luna", "context_length": 1050000, "pricing": { "prompt": "0.00000005", "completion": "0.0000003", "web_search": "0.01", "input_cache_read": "0.000000005", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000001", "completion": "0.00000045", "input_cache_read": "0.00000001" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-luna:batch@openai", "name": "OpenAI: GPT-5.6 Luna", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-luna:batch", "canonicalSlug": "openai/gpt-5.6-luna-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-luna-20260709:batch", "model_id": "openai/gpt-5.6-luna:batch", "model_name": "OpenAI: GPT-5.6 Luna", "context_length": 1050000, "pricing": { "prompt": "0.0000002", "completion": "0.0000012", "web_search": "0.01", "input_cache_read": "0.00000002", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra-pro@openai", "name": "OpenAI: GPT-5.6 Terra Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra-pro", "canonicalSlug": "openai/gpt-5.6-terra-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-pro-20260709", "model_id": "openai/gpt-5.6-terra-pro", "model_name": "OpenAI: GPT-5.6 Terra Pro", "context_length": 1050000, "pricing": { "prompt": "0.000002", "completion": "0.000012", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000004", "completion": "0.000018", "input_cache_read": "0.0000004", "input_cache_write": "0.000005" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9138417659467, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra-pro@openai", "name": "OpenAI: GPT-5.6 Terra Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra-pro", "canonicalSlug": "openai/gpt-5.6-terra-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-pro-20260709", "model_id": "openai/gpt-5.6-terra-pro", "model_name": "OpenAI: GPT-5.6 Terra Pro", "context_length": 1050000, "pricing": { "prompt": "0.000001", "completion": "0.000006", "web_search": "0.01", "input_cache_read": "0.0000001", "input_cache_write": "0.00000125", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000002", "completion": "0.000009", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9138417659467, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra-pro@openai", "name": "OpenAI: GPT-5.6 Terra Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheWrite": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra-pro", "canonicalSlug": "openai/gpt-5.6-terra-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-pro-20260709", "model_id": "openai/gpt-5.6-terra-pro", "model_name": "OpenAI: GPT-5.6 Terra Pro", "context_length": 1050000, "pricing": { "prompt": "0.000004", "completion": "0.000024", "web_search": "0.01", "input_cache_read": "0.0000004", "input_cache_write": "0.000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9138417659467, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra-pro@azure", "name": "OpenAI: GPT-5.6 Terra Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra-pro", "canonicalSlug": "openai/gpt-5.6-terra-pro-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-terra-pro-20260709", "model_id": "openai/gpt-5.6-terra-pro", "model_name": "OpenAI: GPT-5.6 Terra Pro", "context_length": 1050000, "pricing": { "prompt": "0.000002", "completion": "0.000012", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000004", "completion": "0.000018", "input_cache_read": "0.0000004", "input_cache_write": "0.000005" } ] }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.73488865323435, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra-pro@azure", "name": "OpenAI: GPT-5.6 Terra Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 13.200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000132" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra-pro", "canonicalSlug": "openai/gpt-5.6-terra-pro-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-terra-pro-20260709", "model_id": "openai/gpt-5.6-terra-pro", "model_name": "OpenAI: GPT-5.6 Terra Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000022", "completion": "0.0000132", "web_search": "0.01", "input_cache_read": "0.00000022", "input_cache_write": "0.00000275", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000044", "completion": "0.0000198", "input_cache_read": "0.00000044", "input_cache_write": "0.0000055" } ] }, "provider_name": "Azure", "tag": "azure/eu", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra-pro:batch@openai", "name": "OpenAI: GPT-5.6 Terra Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra-pro:batch", "canonicalSlug": "openai/gpt-5.6-terra-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-pro-20260709:batch", "model_id": "openai/gpt-5.6-terra-pro:batch", "model_name": "OpenAI: GPT-5.6 Terra Pro", "context_length": 1050000, "pricing": { "prompt": "0.000001", "completion": "0.000006", "web_search": "0.01", "input_cache_read": "0.0000001", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000002", "completion": "0.000009", "input_cache_read": "0.0000002" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra-pro:batch@openai", "name": "OpenAI: GPT-5.6 Terra Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra-pro:batch", "canonicalSlug": "openai/gpt-5.6-terra-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-pro-20260709:batch", "model_id": "openai/gpt-5.6-terra-pro:batch", "model_name": "OpenAI: GPT-5.6 Terra Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "web_search": "0.01", "input_cache_read": "0.00000005", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000001", "completion": "0.0000045", "input_cache_read": "0.0000001" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra-pro:batch@openai", "name": "OpenAI: GPT-5.6 Terra Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra-pro:batch", "canonicalSlug": "openai/gpt-5.6-terra-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-pro-20260709:batch", "model_id": "openai/gpt-5.6-terra-pro:batch", "model_name": "OpenAI: GPT-5.6 Terra Pro", "context_length": 1050000, "pricing": { "prompt": "0.000002", "completion": "0.000012", "web_search": "0.01", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra@azure", "name": "OpenAI: GPT-5.6 Terra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra", "canonicalSlug": "openai/gpt-5.6-terra-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "uptimeLast30m": 99.91885312415472, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-terra-20260709", "model_id": "openai/gpt-5.6-terra", "model_name": "OpenAI: GPT-5.6 Terra", "context_length": 1050000, "pricing": { "prompt": "0.000002", "completion": "0.000012", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000004", "completion": "0.000018", "input_cache_read": "0.0000004", "input_cache_write": "0.000005" } ] }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.91885312415472, "uptime_last_5m": 100, "uptime_last_1d": 98.99321078745591, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra@openai", "name": "OpenAI: GPT-5.6 Terra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra", "canonicalSlug": "openai/gpt-5.6-terra-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "uptimeLast30m": 99.9465794130413, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-20260709", "model_id": "openai/gpt-5.6-terra", "model_name": "OpenAI: GPT-5.6 Terra", "context_length": 1050000, "pricing": { "prompt": "0.000002", "completion": "0.000012", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000004", "completion": "0.000018", "input_cache_read": "0.0000004", "input_cache_write": "0.000005" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.9465794130413, "uptime_last_5m": 99.95238095238095, "uptime_last_1d": 99.91891676856972, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra@openai", "name": "OpenAI: GPT-5.6 Terra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra", "canonicalSlug": "openai/gpt-5.6-terra-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "uptimeLast30m": 99.9465794130413, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-20260709", "model_id": "openai/gpt-5.6-terra", "model_name": "OpenAI: GPT-5.6 Terra", "context_length": 1050000, "pricing": { "prompt": "0.000001", "completion": "0.000006", "web_search": "0.01", "input_cache_read": "0.0000001", "input_cache_write": "0.00000125", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000002", "completion": "0.000009", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.9465794130413, "uptime_last_5m": 99.95238095238095, "uptime_last_1d": 99.91891676856972, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra@openai", "name": "OpenAI: GPT-5.6 Terra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheWrite": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra", "canonicalSlug": "openai/gpt-5.6-terra-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "uptimeLast30m": 99.9465794130413, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-20260709", "model_id": "openai/gpt-5.6-terra", "model_name": "OpenAI: GPT-5.6 Terra", "context_length": 1050000, "pricing": { "prompt": "0.000004", "completion": "0.000024", "web_search": "0.01", "input_cache_read": "0.0000004", "input_cache_write": "0.000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.9465794130413, "uptime_last_5m": 99.95238095238095, "uptime_last_1d": 99.91891676856972, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra@azure", "name": "OpenAI: GPT-5.6 Terra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 13.200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000132" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra", "canonicalSlug": "openai/gpt-5.6-terra-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-terra-20260709", "model_id": "openai/gpt-5.6-terra", "model_name": "OpenAI: GPT-5.6 Terra", "context_length": 1050000, "pricing": { "prompt": "0.0000022", "completion": "0.0000132", "web_search": "0.01", "input_cache_read": "0.00000022", "input_cache_write": "0.00000275", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000044", "completion": "0.0000198", "input_cache_read": "0.00000044", "input_cache_write": "0.0000055" } ] }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 78.97237154644331, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra@azure", "name": "OpenAI: GPT-5.6 Terra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 13.200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000132" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra", "canonicalSlug": "openai/gpt-5.6-terra-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-terra-20260709", "model_id": "openai/gpt-5.6-terra", "model_name": "OpenAI: GPT-5.6 Terra", "context_length": 1050000, "pricing": { "prompt": "0.0000022", "completion": "0.0000132", "web_search": "0.01", "input_cache_read": "0.00000022", "input_cache_write": "0.00000275", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000044", "completion": "0.0000198", "input_cache_read": "0.00000044", "input_cache_write": "0.0000055" } ] }, "provider_name": "Azure", "tag": "azure/eu", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra@amazon-bedrock", "name": "OpenAI: GPT-5.6 Terra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 13.200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000132" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra", "canonicalSlug": "openai/gpt-5.6-terra-20260709", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | openai/gpt-5.6-terra-20260709", "model_id": "openai/gpt-5.6-terra", "model_name": "OpenAI: GPT-5.6 Terra", "context_length": 1050000, "pricing": { "prompt": "0.0000022", "completion": "0.0000132", "web_search": "0.01", "input_cache_read": "0.00000022", "input_cache_write": "0.00000275", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000044", "completion": "0.0000198", "input_cache_read": "0.00000044", "input_cache_write": "0.0000055" } ] }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/us-east-1", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra:batch@openai", "name": "OpenAI: GPT-5.6 Terra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra:batch", "canonicalSlug": "openai/gpt-5.6-terra-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-20260709:batch", "model_id": "openai/gpt-5.6-terra:batch", "model_name": "OpenAI: GPT-5.6 Terra", "context_length": 1050000, "pricing": { "prompt": "0.000001", "completion": "0.000006", "web_search": "0.01", "input_cache_read": "0.0000001", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000002", "completion": "0.000009", "input_cache_read": "0.0000002" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra:batch@openai", "name": "OpenAI: GPT-5.6 Terra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra:batch", "canonicalSlug": "openai/gpt-5.6-terra-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-20260709:batch", "model_id": "openai/gpt-5.6-terra:batch", "model_name": "OpenAI: GPT-5.6 Terra", "context_length": 1050000, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "web_search": "0.01", "input_cache_read": "0.00000005", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000001", "completion": "0.0000045", "input_cache_read": "0.0000001" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-terra:batch@openai", "name": "OpenAI: GPT-5.6 Terra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-terra:batch", "canonicalSlug": "openai/gpt-5.6-terra-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-terra-20260709:batch", "model_id": "openai/gpt-5.6-terra:batch", "model_name": "OpenAI: GPT-5.6 Terra", "context_length": 1050000, "pricing": { "prompt": "0.000002", "completion": "0.000012", "web_search": "0.01", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol-pro@openai", "name": "OpenAI: GPT-5.6 Sol Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [ { "amount": 3.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol-pro", "canonicalSlug": "openai/gpt-5.6-sol-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": 99.8984771573604, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-pro-20260709", "model_id": "openai/gpt-5.6-sol-pro", "model_name": "OpenAI: GPT-5.6 Sol Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "discount": 0.5, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000005", "completion": "0.0000225", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.8984771573604, "uptime_last_5m": 100, "uptime_last_1d": 99.70822542045045, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol-pro@openai", "name": "OpenAI: GPT-5.6 Sol Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 1.5625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol-pro", "canonicalSlug": "openai/gpt-5.6-sol-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": 99.8984771573604, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-pro-20260709", "model_id": "openai/gpt-5.6-sol-pro", "model_name": "OpenAI: GPT-5.6 Sol Pro", "context_length": 1050000, "pricing": { "prompt": "0.00000125", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.000000125", "input_cache_write": "0.0000015625", "discount": 0.5, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000025", "completion": "0.00001125", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.8984771573604, "uptime_last_5m": 100, "uptime_last_1d": 99.70822542045045, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol-pro@openai", "name": "OpenAI: GPT-5.6 Sol Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol-pro", "canonicalSlug": "openai/gpt-5.6-sol-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": 99.8984771573604, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-pro-20260709", "model_id": "openai/gpt-5.6-sol-pro", "model_name": "OpenAI: GPT-5.6 Sol Pro", "context_length": 1050000, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "discount": 0.5 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.8984771573604, "uptime_last_5m": 100, "uptime_last_1d": 99.70822542045045, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol-pro@azure", "name": "OpenAI: GPT-5.6 Sol Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol-pro", "canonicalSlug": "openai/gpt-5.6-sol-pro-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-sol-pro-20260709", "model_id": "openai/gpt-5.6-sol-pro", "model_name": "OpenAI: GPT-5.6 Sol Pro", "context_length": 1050000, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00001", "completion": "0.000045", "input_cache_read": "0.000001", "input_cache_write": "0.0000125" } ] }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.87410826689047, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol-pro@azure", "name": "OpenAI: GPT-5.6 Sol Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol-pro", "canonicalSlug": "openai/gpt-5.6-sol-pro-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-sol-pro-20260709", "model_id": "openai/gpt-5.6-sol-pro", "model_name": "OpenAI: GPT-5.6 Sol Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000055", "completion": "0.000033", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000011", "completion": "0.0000495", "input_cache_read": "0.0000011", "input_cache_write": "0.00001375" } ] }, "provider_name": "Azure", "tag": "azure/eu", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol-pro:batch@openai", "name": "OpenAI: GPT-5.6 Sol Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol-pro:batch", "canonicalSlug": "openai/gpt-5.6-sol-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-pro-20260709:batch", "model_id": "openai/gpt-5.6-sol-pro:batch", "model_name": "OpenAI: GPT-5.6 Sol Pro", "context_length": 1050000, "pricing": { "prompt": "0.00000125", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.000000125", "discount": 0.5, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000025", "completion": "0.00001125", "input_cache_read": "0.00000025" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol-pro:batch@openai", "name": "OpenAI: GPT-5.6 Sol Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol-pro:batch", "canonicalSlug": "openai/gpt-5.6-sol-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-pro-20260709:batch", "model_id": "openai/gpt-5.6-sol-pro:batch", "model_name": "OpenAI: GPT-5.6 Sol Pro", "context_length": 1050000, "pricing": { "prompt": "0.000000625", "completion": "0.00000375", "web_search": "0.01", "input_cache_read": "0.0000000625", "discount": 0.5, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00000125", "completion": "0.000005625", "input_cache_read": "0.000000125" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol-pro:batch@openai", "name": "OpenAI: GPT-5.6 Sol Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol-pro:batch", "canonicalSlug": "openai/gpt-5.6-sol-pro-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-pro-20260709:batch", "model_id": "openai/gpt-5.6-sol-pro:batch", "model_name": "OpenAI: GPT-5.6 Sol Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0.5 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol@openai", "name": "OpenAI: GPT-5.6 Sol", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [ { "amount": 3.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol", "canonicalSlug": "openai/gpt-5.6-sol-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "uptimeLast30m": 99.78639430762048, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-20260709", "model_id": "openai/gpt-5.6-sol", "model_name": "OpenAI: GPT-5.6 Sol", "context_length": 1050000, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "discount": 0.5, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000005", "completion": "0.0000225", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.78639430762048, "uptime_last_5m": 99.84970382813191, "uptime_last_1d": 99.37081412838539, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol@openai", "name": "OpenAI: GPT-5.6 Sol", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 1.5625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol", "canonicalSlug": "openai/gpt-5.6-sol-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "uptimeLast30m": 99.78639430762048, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-20260709", "model_id": "openai/gpt-5.6-sol", "model_name": "OpenAI: GPT-5.6 Sol", "context_length": 1050000, "pricing": { "prompt": "0.00000125", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.000000125", "input_cache_write": "0.0000015625", "discount": 0.5, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000025", "completion": "0.00001125", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.78639430762048, "uptime_last_5m": 99.84970382813191, "uptime_last_1d": 99.37081412838539, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol@openai", "name": "OpenAI: GPT-5.6 Sol", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol", "canonicalSlug": "openai/gpt-5.6-sol-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "uptimeLast30m": 99.78639430762048, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-20260709", "model_id": "openai/gpt-5.6-sol", "model_name": "OpenAI: GPT-5.6 Sol", "context_length": 1050000, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "discount": 0.5 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.78639430762048, "uptime_last_5m": 99.84970382813191, "uptime_last_1d": 99.37081412838539, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol@azure", "name": "OpenAI: GPT-5.6 Sol", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol", "canonicalSlug": "openai/gpt-5.6-sol-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "uptimeLast30m": 99.91368148467846, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-sol-20260709", "model_id": "openai/gpt-5.6-sol", "model_name": "OpenAI: GPT-5.6 Sol", "context_length": 1050000, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00001", "completion": "0.000045", "input_cache_read": "0.000001", "input_cache_write": "0.0000125" } ] }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.91368148467846, "uptime_last_5m": 99.78813559322035, "uptime_last_1d": 99.15364031529622, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol@azure", "name": "OpenAI: GPT-5.6 Sol", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol", "canonicalSlug": "openai/gpt-5.6-sol-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": -2, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "uptimeLast30m": 91.23460326553996, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-sol-20260709", "model_id": "openai/gpt-5.6-sol", "model_name": "OpenAI: GPT-5.6 Sol", "context_length": 1050000, "pricing": { "prompt": "0.0000055", "completion": "0.000033", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000011", "completion": "0.0000495", "input_cache_read": "0.0000011", "input_cache_write": "0.00001375" } ] }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": -2, "uptime_last_30m": 91.23460326553996, "uptime_last_5m": 100, "uptime_last_1d": 89.8401105118521, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol@azure", "name": "OpenAI: GPT-5.6 Sol", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol", "canonicalSlug": "openai/gpt-5.6-sol-20260709", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.6-sol-20260709", "model_id": "openai/gpt-5.6-sol", "model_name": "OpenAI: GPT-5.6 Sol", "context_length": 1050000, "pricing": { "prompt": "0.0000055", "completion": "0.000033", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000011", "completion": "0.0000495", "input_cache_read": "0.0000011", "input_cache_write": "0.00001375" } ] }, "provider_name": "Azure", "tag": "azure/eu", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.22077922077922, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol@amazon-bedrock", "name": "OpenAI: GPT-5.6 Sol", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol", "canonicalSlug": "openai/gpt-5.6-sol-20260709", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | openai/gpt-5.6-sol-20260709", "model_id": "openai/gpt-5.6-sol", "model_name": "OpenAI: GPT-5.6 Sol", "context_length": 1050000, "pricing": { "prompt": "0.0000055", "completion": "0.000033", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000011", "completion": "0.0000495", "input_cache_read": "0.0000011", "input_cache_write": "0.00001375" } ] }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/us-east-1", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol:batch@openai", "name": "OpenAI: GPT-5.6 Sol", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol:batch", "canonicalSlug": "openai/gpt-5.6-sol-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-20260709:batch", "model_id": "openai/gpt-5.6-sol:batch", "model_name": "OpenAI: GPT-5.6 Sol", "context_length": 1050000, "pricing": { "prompt": "0.00000125", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.000000125", "discount": 0.5, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000025", "completion": "0.00001125", "input_cache_read": "0.00000025" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol:batch@openai", "name": "OpenAI: GPT-5.6 Sol", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol:batch", "canonicalSlug": "openai/gpt-5.6-sol-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-20260709:batch", "model_id": "openai/gpt-5.6-sol:batch", "model_name": "OpenAI: GPT-5.6 Sol", "context_length": 1050000, "pricing": { "prompt": "0.000000625", "completion": "0.00000375", "web_search": "0.01", "input_cache_read": "0.0000000625", "discount": 0.5, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00000125", "completion": "0.000005625", "input_cache_read": "0.000000125" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.6-sol:batch@openai", "name": "OpenAI: GPT-5.6 Sol", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.6-sol:batch", "canonicalSlug": "openai/gpt-5.6-sol-20260709", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.6-sol-20260709:batch", "model_id": "openai/gpt-5.6-sol:batch", "model_name": "OpenAI: GPT-5.6 Sol", "context_length": 1050000, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0.5 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.5@xai", "name": "SpaceXAI: Grok 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.5", "canonicalSlug": "x-ai/grok-4.5-20260708", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 500000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.", "uptimeLast30m": 99.86924306261804, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.5-20260708", "model_id": "x-ai/grok-4.5", "model_name": "SpaceXAI: Grok 4.5", "context_length": 500000, "pricing": { "prompt": "0.000002", "completion": "0.000006", "web_search": "0.005", "input_cache_read": "0.0000003", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000012", "input_cache_read": "0.0000006" } ] }, "provider_name": "xAI", "tag": "xai/zdr", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.86924306261804, "uptime_last_5m": 99.89094874591058, "uptime_last_1d": 99.91340796741855, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.5@xai", "name": "SpaceXAI: Grok 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.5", "canonicalSlug": "x-ai/grok-4.5-20260708", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 500000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.", "uptimeLast30m": 99.86924306261804, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.5-20260708", "model_id": "x-ai/grok-4.5", "model_name": "SpaceXAI: Grok 4.5", "context_length": 500000, "pricing": { "prompt": "0.000004", "completion": "0.000012", "web_search": "0.005", "input_cache_read": "0.0000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000008", "completion": "0.000024", "input_cache_read": "0.0000012" } ] }, "provider_name": "xAI", "tag": "xai/zdr/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.86924306261804, "uptime_last_5m": 99.89094874591058, "uptime_last_1d": 99.91340796741855, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.5@xai", "name": "SpaceXAI: Grok 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.5", "canonicalSlug": "x-ai/grok-4.5-20260708", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 500000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.", "uptimeLast30m": 99.71181556195965, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.5-20260708", "model_id": "x-ai/grok-4.5", "model_name": "SpaceXAI: Grok 4.5", "context_length": 500000, "pricing": { "prompt": "0.000002", "completion": "0.000006", "web_search": "0.005", "input_cache_read": "0.0000003", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000012", "input_cache_read": "0.0000006" } ] }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.71181556195965, "uptime_last_5m": 99.32614555256065, "uptime_last_1d": 99.80228093923073, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.5@xai", "name": "SpaceXAI: Grok 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.5", "canonicalSlug": "x-ai/grok-4.5-20260708", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 500000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.", "uptimeLast30m": 99.71181556195965, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.5-20260708", "model_id": "x-ai/grok-4.5", "model_name": "SpaceXAI: Grok 4.5", "context_length": 500000, "pricing": { "prompt": "0.000004", "completion": "0.000012", "web_search": "0.005", "input_cache_read": "0.0000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000008", "completion": "0.000024", "input_cache_read": "0.0000012" } ] }, "provider_name": "xAI", "tag": "xai/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.71181556195965, "uptime_last_5m": 99.32614555256065, "uptime_last_1d": 99.80228093923073, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "~x-ai/grok-latest", "name": "xAI: Grok Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "canonicalSlug": "~x-ai/grok-latest", "contextLength": 500000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "include_reasoning", "logprobs", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "temperature", "tool_choice", "tools", "top_logprobs", "top_p" ], "description": "This model always redirects to the latest Grok model from xAI.", "endpointCount": 0 } }, { "id": "aion-labs/aion-3.0-mini@aionlabs", "name": "AionLabs: Aion-3.0-Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007" } ], "output": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "cacheRead": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "aion-labs/aion-3.0-mini", "canonicalSlug": "aion-labs/aion-3.0-mini-20260707", "servingProvider": "AionLabs", "servingProviderSlug": "aionlabs", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AionLabs | aion-labs/aion-3.0-mini-20260707", "model_id": "aion-labs/aion-3.0-mini", "model_name": "AionLabs: Aion-3.0-Mini", "context_length": 131072, "pricing": { "prompt": "0.0000007", "completion": "0.0000014", "input_cache_read": "0.00000018", "discount": 0 }, "provider_name": "AionLabs", "tag": "aion-labs", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "aion-labs/aion-3.0@aionlabs", "name": "AionLabs: Aion-3.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "aion-labs/aion-3.0", "canonicalSlug": "aion-labs/aion-3.0-20260707", "servingProvider": "AionLabs", "servingProviderSlug": "aionlabs", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AionLabs | aion-labs/aion-3.0-20260707", "model_id": "aion-labs/aion-3.0", "model_name": "AionLabs: Aion-3.0", "context_length": 131072, "pricing": { "prompt": "0.000003", "completion": "0.000006", "input_cache_read": "0.00000075", "discount": 0 }, "provider_name": "AionLabs", "tag": "aion-labs", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.4781245831666, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "tencent/hy3@gmicloud", "name": "Tencent: Hy3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.126, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000126" } ], "output": [ { "amount": 0.522, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000522" } ], "cacheRead": [ { "amount": 0.0315, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000315" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "tencent/hy3", "canonicalSlug": "tencent/hy3-20260706", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 262144, "quantization": "bf16", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...", "uptimeLast30m": 88.46830985915493, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | tencent/hy3-20260706", "model_id": "tencent/hy3", "model_name": "Tencent: Hy3", "context_length": 262144, "pricing": { "prompt": "0.000000126", "completion": "0.000000522", "input_cache_read": "0.0000000315", "discount": 0.1 }, "provider_name": "GMICloud", "tag": "gmicloud/bf16", "quantization": "bf16", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "reasoning_effort" ], "status": -2, "uptime_last_30m": 88.46830985915493, "uptime_last_5m": 77.02702702702703, "uptime_last_1d": 95.67516861019911, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "tencent/hy3@tencent", "name": "Tencent: Hy3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000132" } ], "output": [ { "amount": 0.5279999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000528" } ], "cacheRead": [ { "amount": 0.032999999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000033" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "tencent/hy3", "canonicalSlug": "tencent/hy3-20260706", "servingProvider": "Tencent", "servingProviderSlug": "tencent", "contextLength": 262144, "maxCompletionTokens": 128000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "stop", "max_completion_tokens", "max_tokens", "response_format", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...", "uptimeLast30m": 96.90745441957074, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Tencent | tencent/hy3-20260706", "model_id": "tencent/hy3", "model_name": "Tencent: Hy3", "context_length": 262144, "pricing": { "prompt": "0.000000132", "completion": "0.000000528", "input_cache_read": "0.000000033", "discount": 0 }, "provider_name": "Tencent", "tag": "tencent/fp8", "quantization": "fp8", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "stop", "max_completion_tokens", "max_tokens", "response_format", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 96.90745441957074, "uptime_last_5m": 96.12903225806451, "uptime_last_1d": 98.7815841248554, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "tencent/hy3@baidu", "name": "Tencent: Hy3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000132" } ], "output": [ { "amount": 0.5279999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000528" } ], "cacheRead": [ { "amount": 0.032999999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000033" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "tencent/hy3", "canonicalSlug": "tencent/hy3-20260706", "servingProvider": "Baidu", "servingProviderSlug": "baidu", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...", "uptimeLast30m": 86.7303609341826, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Baidu | tencent/hy3-20260706", "model_id": "tencent/hy3", "model_name": "Tencent: Hy3", "context_length": 262144, "pricing": { "prompt": "0.000000132", "completion": "0.000000528", "input_cache_read": "0.000000033", "discount": 0 }, "provider_name": "Baidu", "tag": "baidu/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": -2, "uptime_last_30m": 86.7303609341826, "uptime_last_5m": 84.61538461538461, "uptime_last_1d": 97.05513927164233, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "tencent/hy3@deepinfra", "name": "Tencent: Hy3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.58, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000058" } ], "cacheRead": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000035" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "tencent/hy3", "canonicalSlug": "tencent/hy3-20260706", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...", "uptimeLast30m": 99.37106918238993, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | tencent/hy3-20260706", "model_id": "tencent/hy3", "model_name": "Tencent: Hy3", "context_length": 262144, "pricing": { "prompt": "0.00000014", "completion": "0.00000058", "input_cache_read": "0.000000035", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.37106918238993, "uptime_last_5m": 100, "uptime_last_1d": 98.85209713024283, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "tencent/hy3@novita", "name": "Tencent: Hy3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.58, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000058" } ], "cacheRead": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000035" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "tencent/hy3", "canonicalSlug": "tencent/hy3-20260706", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...", "uptimeLast30m": 87.16861081654295, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | tencent/hy3-20260706", "model_id": "tencent/hy3", "model_name": "Tencent: Hy3", "context_length": 262144, "pricing": { "prompt": "0.00000014", "completion": "0.00000058", "input_cache_read": "0.000000035", "discount": 0 }, "provider_name": "Novita", "tag": "novita", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": -2, "uptime_last_30m": 87.16861081654295, "uptime_last_5m": 80, "uptime_last_1d": 97.93944765206511, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "tencent/hy3@atlascloud", "name": "Tencent: Hy3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "tencent/hy3", "canonicalSlug": "tencent/hy3-20260706", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "fp8", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...", "uptimeLast30m": 83.01282051282051, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | tencent/hy3-20260706", "model_id": "tencent/hy3", "model_name": "Tencent: Hy3", "context_length": 262144, "pricing": { "prompt": "0.0000002", "completion": "0.0000008", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": -2, "uptime_last_30m": 83.01282051282051, "uptime_last_5m": null, "uptime_last_1d": 91.75875729774813, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "poolside/laguna-xs-2.1@poolside", "name": "Poolside: Laguna XS 2.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "output": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "poolside/laguna-xs-2.1", "canonicalSlug": "poolside/laguna-xs-2.1-20260625", "servingProvider": "Poolside", "servingProviderSlug": "poolside", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Poolside | poolside/laguna-xs-2.1-20260625", "model_id": "poolside/laguna-xs-2.1", "model_name": "Poolside: Laguna XS 2.1", "context_length": 262144, "pricing": { "prompt": "0.00000006", "completion": "0.00000012", "input_cache_read": "0.00000003", "discount": 0.4 }, "provider_name": "Poolside", "tag": "poolside/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "poolside/laguna-xs-2.1:free@poolside", "name": "Poolside: Laguna XS 2.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "poolside/laguna-xs-2.1:free", "canonicalSlug": "poolside/laguna-xs-2.1-20260625", "servingProvider": "Poolside", "servingProviderSlug": "poolside", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...", "uptimeLast30m": 99.98247458815283, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Poolside | poolside/laguna-xs-2.1-20260625:free", "model_id": "poolside/laguna-xs-2.1:free", "model_name": "Poolside: Laguna XS 2.1", "context_length": 262144, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Poolside", "tag": "poolside/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.98247458815283, "uptime_last_5m": 100, "uptime_last_1d": 99.94499779242841, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-5@claude-platform-on-aws", "name": "Anthropic: Claude Sonnet 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-5", "canonicalSlug": "anthropic/claude-sonnet-5-20260630", "servingProvider": "Claude Platform on AWS", "servingProviderSlug": "claude-platform-on-aws", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "uptimeLast30m": 99.94644335167243, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Claude Platform on AWS | anthropic/claude-sonnet-5-20260630", "model_id": "anthropic/claude-sonnet-5", "model_name": "Anthropic: Claude Sonnet 5", "context_length": 1000000, "pricing": { "prompt": "0.000002", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "input_cache_write_1h": "0.000004", "discount": 0 }, "provider_name": "Claude Platform on AWS", "tag": "claude-on-aws", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.94644335167243, "uptime_last_5m": 100, "uptime_last_1d": 99.90503219668324, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-5@azure", "name": "Anthropic: Claude Sonnet 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-5", "canonicalSlug": "anthropic/claude-sonnet-5-20260630", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-sonnet-5-20260630", "model_id": "anthropic/claude-sonnet-5", "model_name": "Anthropic: Claude Sonnet 5", "context_length": 1000000, "pricing": { "prompt": "0.000002", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "input_cache_write_1h": "0.000004", "discount": 0 }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.9834665847869, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-5@anthropic", "name": "Anthropic: Claude Sonnet 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-5", "canonicalSlug": "anthropic/claude-sonnet-5-20260630", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-sonnet-5-20260630", "model_id": "anthropic/claude-sonnet-5", "model_name": "Anthropic: Claude Sonnet 5", "context_length": 1000000, "pricing": { "prompt": "0.000002", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "input_cache_write_1h": "0.000004", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.95044296676024, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-5@amazon-bedrock", "name": "Anthropic: Claude Sonnet 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-5", "canonicalSlug": "anthropic/claude-sonnet-5-20260630", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "uptimeLast30m": 99.9599077879122, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-sonnet-5-20260630", "model_id": "anthropic/claude-sonnet-5", "model_name": "Anthropic: Claude Sonnet 5", "context_length": 1000000, "pricing": { "prompt": "0.000002", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "input_cache_write_1h": "0.000004", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.9599077879122, "uptime_last_5m": 100, "uptime_last_1d": 97.88654039843941, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-5@google", "name": "Anthropic: Claude Sonnet 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-5", "canonicalSlug": "anthropic/claude-sonnet-5-20260630", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-sonnet-5-20260630", "model_id": "anthropic/claude-sonnet-5", "model_name": "Anthropic: Claude Sonnet 5", "context_length": 1000000, "pricing": { "prompt": "0.000002", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "input_cache_write_1h": "0.000004", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.88517356865664, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-5@azure", "name": "Anthropic: Claude Sonnet 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-5", "canonicalSlug": "anthropic/claude-sonnet-5-20260630", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-sonnet-5-20260630", "model_id": "anthropic/claude-sonnet-5", "model_name": "Anthropic: Claude Sonnet 5", "context_length": 1000000, "pricing": { "prompt": "0.000002", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.0000002", "input_cache_write": "0.0000025", "input_cache_write_1h": "0.000004", "discount": 0 }, "provider_name": "Azure", "tag": "azure/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.57789382071365, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-5@amazon-bedrock", "name": "Anthropic: Claude Sonnet 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-5", "canonicalSlug": "anthropic/claude-sonnet-5-20260630", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-sonnet-5-20260630", "model_id": "anthropic/claude-sonnet-5", "model_name": "Anthropic: Claude Sonnet 5", "context_length": 1000000, "pricing": { "prompt": "0.0000022", "completion": "0.000011", "web_search": "0.01", "input_cache_read": "0.00000022", "input_cache_write": "0.00000275", "input_cache_write_1h": "0.0000044", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/us-east-1", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-5@google", "name": "Anthropic: Claude Sonnet 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-5", "canonicalSlug": "anthropic/claude-sonnet-5-20260630", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-sonnet-5-20260630", "model_id": "anthropic/claude-sonnet-5", "model_name": "Anthropic: Claude Sonnet 5", "context_length": 1000000, "pricing": { "prompt": "0.0000022", "completion": "0.000011", "web_search": "0.01", "input_cache_read": "0.00000022", "input_cache_write": "0.00000275", "input_cache_write_1h": "0.0000044", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/europe", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.98727330575883, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-5@google", "name": "Anthropic: Claude Sonnet 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-5", "canonicalSlug": "anthropic/claude-sonnet-5-20260630", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-sonnet-5-20260630", "model_id": "anthropic/claude-sonnet-5", "model_name": "Anthropic: Claude Sonnet 5", "context_length": 1000000, "pricing": { "prompt": "0.0000022", "completion": "0.000011", "web_search": "0.01", "input_cache_read": "0.00000022", "input_cache_write": "0.00000275", "input_cache_write_1h": "0.0000044", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-5:batch@anthropic", "name": "Anthropic: Claude Sonnet 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-5:batch", "canonicalSlug": "anthropic/claude-sonnet-5-20260630", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-sonnet-5-20260630:batch", "model_id": "anthropic/claude-sonnet-5:batch", "model_name": "Anthropic: Claude Sonnet 5", "context_length": 1000000, "pricing": { "prompt": "0.000001", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000001", "input_cache_write": "0.00000125", "input_cache_write_1h": "0.000002", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite-image@google", "name": "Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000025" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite-image", "canonicalSlug": "google/gemini-3.1-flash-lite-image-20260630", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 65536, "maxCompletionTokens": 66000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "reasoning_effort" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...", "uptimeLast30m": 99.89200863930886, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-flash-lite-image-20260630", "model_id": "google/gemini-3.1-flash-lite-image", "model_name": "Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)", "context_length": 65536, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image_output": "0.00003", "web_search": "0.014", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 66000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.89200863930886, "uptime_last_5m": 100, "uptime_last_1d": 99.75450323878258, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite-image@google-ai-studio", "name": "Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000025" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite-image", "canonicalSlug": "google/gemini-3.1-flash-lite-image-20260630", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 65536, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "reasoning_effort" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-flash-lite-image-20260630", "model_id": "google/gemini-3.1-flash-lite-image", "model_name": "Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)", "context_length": 65536, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image_output": "0.00003", "web_search": "0.014", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.52855779843975, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nex-agi/nex-n2-mini@nex-agi", "name": "Nex AGI: Nex-N2-Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheRead": [ { "amount": 0.0025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nex-agi/nex-n2-mini", "canonicalSlug": "nex-agi/nex-n2-mini", "servingProvider": "Nex AGI", "servingProviderSlug": "nex-agi", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "top_p", "top_k", "temperature", "max_tokens", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Nex-N2-Mini is an open-source agentic mixture-of-experts model from Nex AGI, the smaller sibling in the Nex-N2 series. It accepts text and image input and is built for coding, tool use,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nex AGI | nex-agi/nex-n2-mini", "model_id": "nex-agi/nex-n2-mini", "model_name": "Nex AGI: Nex-N2-Mini", "context_length": 262144, "pricing": { "prompt": "0.000000025", "completion": "0.0000001", "input_cache_read": "0.0000000025", "discount": 0 }, "provider_name": "Nex AGI", "tag": "nex-agi", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "top_p", "top_k", "temperature", "max_tokens", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.98629234449548, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sakana/fugu-ultra@sakana-ai", "name": "Sakana: Fugu Ultra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "sakana/fugu-ultra", "canonicalSlug": "sakana/fugu-ultra-20260615", "servingProvider": "Sakana AI", "servingProviderSlug": "sakana-ai", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "web_search_options", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Sakana AI | sakana/fugu-ultra-20260615", "model_id": "sakana/fugu-ultra", "model_name": "Sakana: Fugu Ultra", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00001", "completion": "0.000045", "input_cache_read": "0.000001" } ] }, "provider_name": "Sakana AI", "tag": "sakana", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "web_search_options", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.94036970781157, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "alibaba/happyhorse-1.1@alibaba", "name": "Alibaba: HappyHorse 1.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "alibaba/happyhorse-1.1", "canonicalSlug": "alibaba/happyhorse-1.1-20260624", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "HappyHorse 1.1 is a video generation model from Alibaba. It generates short videos from a text prompt, a single starting image, or a set of reference images, with output up...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | alibaba/happyhorse-1.1-20260624", "model_id": "alibaba/happyhorse-1.1", "model_name": "Alibaba: HappyHorse 1.1", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-image-2@openai", "name": "OpenAI: GPT Image 2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "cacheRead": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-image-2", "canonicalSlug": "openai/gpt-image-2", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI's latest image generation model. Supports high-fidelity image generation and editing via the dedicated Images API.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-image-2", "model_id": "openai/gpt-image-2", "model_name": "OpenAI: GPT Image 2", "context_length": 400000, "pricing": { "prompt": "0.000008", "completion": "0.000008", "image_output": "0.00003", "web_search": "0.01", "input_cache_read": "0.000002", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.95476068401847, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-image-1@openai", "name": "OpenAI: GPT Image 1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-image-1", "canonicalSlug": "openai/gpt-image-1", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI's GPT Image 1 generates and edits images via the dedicated Images API. Features accurate text rendering, transparent backgrounds, and up to 16 reference images for edits.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-image-1", "model_id": "openai/gpt-image-1", "model_name": "OpenAI: GPT Image 1", "context_length": 400000, "pricing": { "prompt": "0.00001", "completion": "0.00001", "image_output": "0.00004", "web_search": "0.01", "input_cache_read": "0.00000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-image-1-mini@openai", "name": "OpenAI: GPT Image 1 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-image-1-mini", "canonicalSlug": "openai/gpt-image-1-mini", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "GPT", "instruct_type": null }, "description": "A cost-efficient variant of GPT Image 1 for high-quality image generation at reduced latency and cost via OpenAI's dedicated Images API.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-image-1-mini", "model_id": "openai/gpt-image-1-mini", "model_name": "OpenAI: GPT Image 1 Mini", "context_length": 400000, "pricing": { "prompt": "0.0000025", "completion": "0.0000025", "image_output": "0.000008", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.92576095025983, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "alibaba/happyhorse-1.0@alibaba", "name": "Alibaba: HappyHorse 1.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "alibaba/happyhorse-1.0", "canonicalSlug": "alibaba/happyhorse-1.0-20260624", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "HappyHorse 1.0 is a video generation model from Alibaba. It generates short videos from a text prompt, a single starting image, or a set of reference images, with output up...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | alibaba/happyhorse-1.0-20260624", "model_id": "alibaba/happyhorse-1.0", "model_name": "Alibaba: HappyHorse 1.0", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-image@google", "name": "Google: Nano Banana 2 (Gemini 3.1 Flash Image)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000005" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-image", "canonicalSlug": "google/gemini-3.1-flash-image-20260528", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "reasoning_effort" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...", "uptimeLast30m": 99.37469937469938, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-flash-image-20260528", "model_id": "google/gemini-3.1-flash-image", "model_name": "Google: Nano Banana 2 (Gemini 3.1 Flash Image)", "context_length": 131072, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "image_output": "0.00006", "web_search": "0.014", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.37469937469938, "uptime_last_5m": 99.59183673469387, "uptime_last_1d": 99.38383750997299, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-image@google-ai-studio", "name": "Google: Nano Banana 2 (Gemini 3.1 Flash Image)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000005" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-image", "canonicalSlug": "google/gemini-3.1-flash-image-20260528", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 65536, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...", "uptimeLast30m": 98.2825484764543, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-flash-image-20260528", "model_id": "google/gemini-3.1-flash-image", "model_name": "Google: Nano Banana 2 (Gemini 3.1 Flash Image)", "context_length": 65536, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "image_output": "0.00006", "web_search": "0.014", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.2825484764543, "uptime_last_5m": 98.29545454545455, "uptime_last_1d": 99.12884127431633, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3-pro-image@google", "name": "Google: Nano Banana Pro (Gemini 3 Pro Image)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000000375" } ], "other": [ { "amount": 0.000002, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000002" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3-pro-image", "canonicalSlug": "google/gemini-3-pro-image-20260528", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 65536, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...", "uptimeLast30m": 99.67948717948718, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3-pro-image-20260528", "model_id": "google/gemini-3-pro-image", "model_name": "Google: Nano Banana Pro (Gemini 3 Pro Image)", "context_length": 65536, "pricing": { "prompt": "0.000002", "completion": "0.000012", "image": "0.000002", "image_output": "0.00012", "audio": "0.000002", "input_audio_cache": "0.0000002", "web_search": "0.014", "internal_reasoning": "0.000012", "input_cache_read": "0.0000002", "input_cache_write": "0.000000375", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.67948717948718, "uptime_last_5m": 100, "uptime_last_1d": 99.82173441923871, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3-pro-image@google-ai-studio", "name": "Google: Nano Banana Pro (Gemini 3 Pro Image)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000000375" } ], "other": [ { "amount": 0.000002, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000002" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3-pro-image", "canonicalSlug": "google/gemini-3-pro-image-20260528", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3-pro-image-20260528", "model_id": "google/gemini-3-pro-image", "model_name": "Google: Nano Banana Pro (Gemini 3 Pro Image)", "context_length": 131072, "pricing": { "prompt": "0.000002", "completion": "0.000012", "image": "0.000002", "image_output": "0.00012", "audio": "0.000002", "input_audio_cache": "0.0000002", "web_search": "0.014", "internal_reasoning": "0.000012", "input_cache_read": "0.0000002", "input_cache_write": "0.000000375", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/global", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 98.44000519998268, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "cohere/north-mini-code:free@cohere", "name": "Cohere: North Mini Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "cohere/north-mini-code:free", "canonicalSlug": "cohere/north-mini-code-20260617", "servingProvider": "Cohere", "servingProviderSlug": "cohere", "contextLength": 256000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Cohere", "instruct_type": null }, "description": "North Mini Code is Cohere's first agentic coding model and the debut of its North family. A sparse mixture-of-experts model with 30B total parameters and 3B active, it is optimized...", "uptimeLast30m": 97.81813595661497, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cohere | cohere/north-mini-code-20260617:free", "model_id": "cohere/north-mini-code:free", "model_name": "Cohere: North Mini Code", "context_length": 256000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Cohere", "tag": "cohere", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 97.81813595661497, "uptime_last_5m": 97.47899159663865, "uptime_last_1d": 97.52277631263486, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@sail-research", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 3.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000315" } ], "cacheRead": [ { "amount": 0.11499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000115" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Sail Research", "servingProviderSlug": "sail-research", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "tools", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "logprobs", "top_logprobs", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 96.66615612395977, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Sail Research | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.0000005", "completion": "0.00000315", "input_cache_read": "0.000000115", "discount": 0 }, "provider_name": "Sail Research", "tag": "sail-research/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "tools", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "logprobs", "top_logprobs", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 96.66615612395977, "uptime_last_5m": 94.53057708871663, "uptime_last_1d": 99.56254633434287, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@ambient", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Ambient", "servingProviderSlug": "ambient", "contextLength": 202752, "maxCompletionTokens": 202752, "quantization": "fp8", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "top_p", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 83.45771144278606, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Ambient | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 202752, "pricing": { "prompt": "0.0000006", "completion": "0.000002", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "Ambient", "tag": "ambient/fp8", "quantization": "fp8", "max_completion_tokens": 202752, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "top_p", "reasoning_effort" ], "status": -2, "uptime_last_30m": 83.45771144278606, "uptime_last_5m": 62.5, "uptime_last_1d": 81.60873882820259, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@decart", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.684, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000684" } ], "output": [ { "amount": 2.2800000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000228" } ], "cacheRead": [ { "amount": 0.114, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000114" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Decart", "servingProviderSlug": "decart", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.7679443066336, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Decart | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.000000684", "completion": "0.00000228", "input_cache_read": "0.000000114", "discount": 0.43 }, "provider_name": "Decart", "tag": "decart/fp4", "quantization": "fp4", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.7679443066336, "uptime_last_5m": 99.59677419354838, "uptime_last_1d": 99.28484983894693, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@digitalocean", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.105, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000105" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "tool_choice", "tools", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.56114686951435, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 262144, "pricing": { "prompt": "0.0000007", "completion": "0.0000022", "input_cache_read": "0.000000105", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "tool_choice", "tools", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.56114686951435, "uptime_last_5m": 99.3421052631579, "uptime_last_1d": 98.43231015920284, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@streamlake", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7335999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007336" } ], "output": [ { "amount": 2.3056, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000023056" } ], "cacheRead": [ { "amount": 0.13624, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013624" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 1024000, "maxCompletionTokens": 128000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.92378048780488, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1024000, "pricing": { "prompt": "0.0000007336", "completion": "0.0000023056", "input_cache_read": "0.00000013624", "discount": 0.476 }, "provider_name": "StreamLake", "tag": "streamlake/fp8", "quantization": "fp8", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.92378048780488, "uptime_last_5m": 100, "uptime_last_1d": 99.5943257447622, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@novita", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7405999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007406" } ], "output": [ { "amount": 2.3276, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000023276" } ], "cacheRead": [ { "amount": 0.13754, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013754" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.94544037412315, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.0000007406", "completion": "0.0000023276", "input_cache_read": "0.00000013754", "discount": 0.471 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.94544037412315, "uptime_last_5m": 99.93021632937894, "uptime_last_1d": 98.71729067137122, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@gmicloud", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.742, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000742" } ], "output": [ { "amount": 2.332, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002332" } ], "cacheRead": [ { "amount": 0.1378, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001378" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "structured_outputs", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.64852902889872, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.000000742", "completion": "0.000002332", "input_cache_read": "0.0000001378", "discount": 0.47 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "structured_outputs", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.64852902889872, "uptime_last_5m": 98.90710382513662, "uptime_last_1d": 99.11452026605645, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@deepinfra", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 1048576, "maxCompletionTokens": 163840, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.54098360655738, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.00000075", "completion": "0.0000024", "input_cache_read": "0.00000014", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 163840, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.54098360655738, "uptime_last_5m": 99.37343358395991, "uptime_last_1d": 99.1756081431824, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@inceptron", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 2.9000000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000029" } ], "cacheRead": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000017" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Inceptron", "servingProviderSlug": "inceptron", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp4", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "stop", "frequency_penalty", "logit_bias", "parallel_tool_calls", "presence_penalty", "seed", "response_format", "logprobs", "top_logprobs", "structured_outputs", "tools", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 94.41255262150786, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Inceptron | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.00000075", "completion": "0.0000029", "input_cache_read": "0.00000017", "discount": 0 }, "provider_name": "Inceptron", "tag": "inceptron/fp4", "quantization": "fp4", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "stop", "frequency_penalty", "logit_bias", "parallel_tool_calls", "presence_penalty", "seed", "response_format", "logprobs", "top_logprobs", "structured_outputs", "tools", "reasoning_effort" ], "status": -2, "uptime_last_30m": 94.41255262150786, "uptime_last_5m": 100, "uptime_last_1d": 95.7295255489602, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@coreweave", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.76, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000076" } ], "output": [ { "amount": 2.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000242" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 262144, "pricing": { "prompt": "0.00000076", "completion": "0.00000242", "input_cache_read": "0.00000014", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 90.06898643014176, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@akashml", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.77, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000077" } ], "output": [ { "amount": 2.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000242" } ], "cacheRead": [ { "amount": 0.143, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000143" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "AkashML", "servingProviderSlug": "akashml", "contextLength": 96890, "maxCompletionTokens": 96890, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "tools", "logprobs", "top_logprobs", "response_format", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 97.53761969904241, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AkashML | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 96890, "pricing": { "prompt": "0.00000077", "completion": "0.00000242", "input_cache_read": "0.000000143", "discount": 0.45 }, "provider_name": "AkashML", "tag": "akashml/fp8", "quantization": "fp8", "max_completion_tokens": 96890, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "tools", "logprobs", "top_logprobs", "response_format", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.53761969904241, "uptime_last_5m": 95.37037037037037, "uptime_last_1d": 95.42855376626588, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@alibaba", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.966, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000966" } ], "output": [ { "amount": 3.036, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003036" } ], "cacheRead": [ { "amount": 0.1932, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001932" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.70742268644489, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.000000966", "completion": "0.000003036", "input_cache_read": "0.0000001932", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": 1048576, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.70742268644489, "uptime_last_5m": 99.50267084177565, "uptime_last_1d": 99.62670080428666, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@morph", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000041" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Morph", "servingProviderSlug": "morph", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.40298507462687, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Morph | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.0000011", "completion": "0.0000041", "input_cache_read": "0.00000022", "discount": 0 }, "provider_name": "Morph", "tag": "morph/fp4", "quantization": "fp4", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.40298507462687, "uptime_last_5m": 98.94179894179894, "uptime_last_1d": 95.80554666929153, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@phala", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.1340000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001134" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.21059999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002106" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.92862241256245, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.000001134", "completion": "0.000003", "input_cache_read": "0.0000002106", "discount": 0 }, "provider_name": "Phala", "tag": "phala/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.92862241256245, "uptime_last_5m": 99.82456140350877, "uptime_last_1d": 99.55927795411009, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@siliconflow", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000119" } ], "output": [ { "amount": 3.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000374" } ], "cacheRead": [ { "amount": 0.221, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000221" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 1048576, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.94403402731139, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.00000119", "completion": "0.00000374", "input_cache_read": "0.000000221", "discount": 0.15 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.94403402731139, "uptime_last_5m": 100, "uptime_last_1d": 99.82376023832529, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@wafer", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000126" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.234, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000234" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Wafer", "servingProviderSlug": "wafer", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "logit_bias", "structured_outputs", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 98.86280264123258, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Wafer | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.00000126", "completion": "0.00000396", "input_cache_read": "0.000000234", "discount": 0 }, "provider_name": "Wafer", "tag": "wafer", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "logit_bias", "structured_outputs", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.86280264123258, "uptime_last_5m": 98.3402489626556, "uptime_last_1d": 98.127026793751, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@atlascloud", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000126" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.234, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000234" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.00000126", "completion": "0.00000396", "input_cache_read": "0.000000234", "discount": 0.1 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 97.25518459492656, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@z.ai", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.89773308334755, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.89773308334755, "uptime_last_5m": 100, "uptime_last_1d": 98.43858641749752, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@fireworks", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Fireworks", "servingProviderSlug": "fireworks", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 97.00061842918986, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fireworks | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000014", "discount": 0 }, "provider_name": "Fireworks", "tag": "fireworks", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.00061842918986, "uptime_last_5m": 95, "uptime_last_1d": 97.93763271538664, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@cloudflare", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "tools", "structured_outputs", "response_format", "logprobs", "top_logprobs", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 262144, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "tools", "structured_outputs", "response_format", "logprobs", "top_logprobs", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.71268954509178, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@friendli", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Friendli", "servingProviderSlug": "friendli", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.86733001658375, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Friendli | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Friendli", "tag": "friendli", "quantization": "unknown", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.86733001658375, "uptime_last_5m": 99.9165971643036, "uptime_last_1d": 99.64750092216894, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@parasail", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.72958355868037, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 262144, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.72958355868037, "uptime_last_5m": 100, "uptime_last_1d": 99.80921731573515, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@venice", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 1000000, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 97.87697332607512, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1000000, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.87697332607512, "uptime_last_5m": 98.2367758186398, "uptime_last_1d": 97.74831172672155, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@together", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 512000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 99.86850756081526, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 512000, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.86850756081526, "uptime_last_5m": 99.68602825745683, "uptime_last_1d": 93.09200180609402, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@baidu", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Baidu", "servingProviderSlug": "baidu", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Baidu | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Baidu", "tag": "baidu/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.67784480335561, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@crusoe", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Crusoe", "servingProviderSlug": "crusoe", "contextLength": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Crusoe | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Crusoe", "tag": "crusoe/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.60105004292237, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@baseten", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "BaseTen", "servingProviderSlug": "baseten", "contextLength": 1048576, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "BaseTen | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000014", "discount": 0 }, "provider_name": "BaseTen", "tag": "baseten/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.95690893421431, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@fireworks", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.0999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000021" } ], "output": [ { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "cacheRead": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000021" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Fireworks", "servingProviderSlug": "fireworks", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 97.70114942528735, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fireworks | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.0000021", "completion": "0.0000066", "input_cache_read": "0.00000021", "discount": 0 }, "provider_name": "Fireworks", "tag": "fireworks/fast", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.70114942528735, "uptime_last_5m": 90.32258064516128, "uptime_last_1d": 91.8482007445195, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@cloudflare", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.0999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000021" } ], "output": [ { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "cacheRead": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000021" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 262144, "pricing": { "prompt": "0.0000021", "completion": "0.0000066", "input_cache_read": "0.00000021", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare/fast", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.79596321059736, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@baseten", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.0999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000021" } ], "output": [ { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "cacheRead": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000021" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "BaseTen", "servingProviderSlug": "baseten", "contextLength": 1048576, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "BaseTen | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.0000021", "completion": "0.0000066", "input_cache_read": "0.00000021", "discount": 0 }, "provider_name": "BaseTen", "tag": "baseten/fast", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.81404958677686, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2@alibaba", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.31, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000231" } ], "output": [ { "amount": 7.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000726" } ], "cacheRead": [ { "amount": 0.46199999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000462" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | z-ai/glm-5.2-20260616", "model_id": "z-ai/glm-5.2", "model_name": "Z.ai: GLM 5.2", "context_length": 1048576, "pricing": { "prompt": "0.00000231", "completion": "0.00000726", "input_cache_read": "0.000000462", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba/fast", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": 1048576, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.85920167348868, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2:batch@together", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2:batch", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 512000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | z-ai/glm-5.2-20260616:batch", "model_id": "z-ai/glm-5.2:batch", "model_name": "Z.ai: GLM 5.2", "context_length": 512000, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.2:free@decart", "name": "Z.ai: GLM 5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.2:free", "canonicalSlug": "z-ai/glm-5.2-20260616", "servingProvider": "Decart", "servingProviderSlug": "decart", "contextLength": 256000, "maxCompletionTokens": 256000, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...", "uptimeLast30m": 98.44377510040161, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Decart | z-ai/glm-5.2-20260616:free", "model_id": "z-ai/glm-5.2:free", "model_name": "Z.ai: GLM 5.2", "context_length": 256000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Decart", "tag": "decart/fp4", "quantization": "fp4", "max_completion_tokens": 256000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.44377510040161, "uptime_last_5m": 99.32885906040269, "uptime_last_1d": 98.84491051013167, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openrouter/fusion", "name": "OpenRouter: Fusion", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": -1000000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "-1" } ], "output": [ { "amount": -1000000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "-1" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "canonicalSlug": "openrouter/fusion", "contextLength": 1000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [], "description": "Fusion turns your prompt into a small multi-model deliberation. A panel of expert models (see below) analyzes your prompt in parallel with web search and web fetch enabled, then a...", "endpointCount": 0 } }, { "id": "moonshotai/kimi-k2.7-code@inceptron", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.67, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000067" } ], "output": [ { "amount": 3.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000034" } ], "cacheRead": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000017" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "Inceptron", "servingProviderSlug": "inceptron", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "frequency_penalty", "logit_bias", "parallel_tool_calls", "presence_penalty", "seed" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Inceptron | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000067", "completion": "0.0000034", "input_cache_read": "0.00000017", "discount": 0 }, "provider_name": "Inceptron", "tag": "inceptron/int4", "quantization": "int4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "frequency_penalty", "logit_bias", "parallel_tool_calls", "presence_penalty", "seed" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.84542900881353, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@deepinfra", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6799999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000068" } ], "output": [ { "amount": 3.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000034" } ], "cacheRead": [ { "amount": 0.136, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000136" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000068", "completion": "0.0000034", "input_cache_read": "0.000000136", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.66358284272498, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@ambient", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.69, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000069" } ], "output": [ { "amount": 3.49, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000349" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "Ambient", "servingProviderSlug": "ambient", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "frequency_penalty", "top_p", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Ambient | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000069", "completion": "0.00000349", "input_cache_read": "0.00000014", "discount": 0 }, "provider_name": "Ambient", "tag": "ambient", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "frequency_penalty", "top_p", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 97.25373134328358, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@coreweave", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.71, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000071" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000071", "completion": "0.0000035", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/int4", "quantization": "int4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.97026843357119, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@venice", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 65536, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "response_format", "structured_outputs", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 256000, "pricing": { "prompt": "0.00000075", "completion": "0.0000035", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "Venice", "tag": "venice/int4", "quantization": "int4", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "response_format", "structured_outputs", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 94.77883538633819, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@parasail", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.76, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000076" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000076", "completion": "0.0000035", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/int4", "quantization": "int4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 97.48415776033292, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@modelrun", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000085" } ], "output": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "ModelRun", "servingProviderSlug": "modelrun", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "presence_penalty", "repetition_penalty", "frequency_penalty", "top_p", "top_k", "stop", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "ModelRun | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000085", "completion": "0.00000375", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "ModelRun", "tag": "modelrun/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "presence_penalty", "repetition_penalty", "frequency_penalty", "top_p", "top_k", "stop", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 97.50885098165433, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@siliconflow", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.85916, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000085916" } ], "output": [ { "amount": 3.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000038" } ], "cacheRead": [ { "amount": 0.17993, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000017993" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000085916", "completion": "0.0000038", "input_cache_read": "0.00000017993", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.94633033677714, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@novita", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.912, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000912" } ], "output": [ { "amount": 3.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000384" } ], "cacheRead": [ { "amount": 0.18239999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001824" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.000000912", "completion": "0.00000384", "input_cache_read": "0.0000001824", "discount": 0.04 }, "provider_name": "Novita", "tag": "novita/int4", "quantization": "int4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.18997107039537, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@moonshot-ai", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000019" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "Moonshot AI", "servingProviderSlug": "moonshot-ai", "contextLength": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "frequency_penalty", "presence_penalty", "structured_outputs", "response_format", "tool_choice", "tools" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Moonshot AI | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000019", "discount": 0 }, "provider_name": "Moonshot AI", "tag": "moonshotai/int4", "quantization": "int4", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "frequency_penalty", "presence_penalty", "structured_outputs", "response_format", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.00467501131051, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@cloudflare", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000019" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000019", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.90409973627428, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@atlascloud", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/int4", "quantization": "int4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.83056590986106, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@together", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000019" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000019", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 93.29858525688756, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@alibaba", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000019" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "structured_outputs", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000019", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": 229376, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "structured_outputs", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.9544142227625, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code@moonshot-ai", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000019" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000038" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "Moonshot AI", "servingProviderSlug": "moonshot-ai", "contextLength": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "frequency_penalty", "presence_penalty", "tools", "response_format", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": 98.48484848484848, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Moonshot AI | moonshotai/kimi-k2.7-code-20260612", "model_id": "moonshotai/kimi-k2.7-code", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.0000019", "completion": "0.000008", "input_cache_read": "0.00000038", "discount": 0 }, "provider_name": "Moonshot AI", "tag": "moonshotai/highspeed", "quantization": "int4", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "frequency_penalty", "presence_penalty", "tools", "response_format", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 98.48484848484848, "uptime_last_5m": null, "uptime_last_1d": 99.95086960793948, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.7-code:batch@together", "name": "MoonshotAI: Kimi K2.7 Code", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000019" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.7-code:batch", "canonicalSlug": "moonshotai/kimi-k2.7-code-20260612", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | moonshotai/kimi-k2.7-code-20260612:batch", "model_id": "moonshotai/kimi-k2.7-code:batch", "model_name": "MoonshotAI: Kimi K2.7 Code", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000019", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/llama-nemotron-rerank-vl-1b-v2:free@nvidia", "name": "NVIDIA: Llama Nemotron Rerank VL 1B V2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/llama-nemotron-rerank-vl-1b-v2:free", "canonicalSlug": "nvidia/llama-nemotron-rerank-vl-1b-v2", "servingProvider": "Nvidia", "servingProviderSlug": "nvidia", "contextLength": 10240, "quantization": "unknown", "status": 0, "supportedParameters": [ "temperature", "max_tokens", "seed", "top_p" ], "architecture": { "modality": "text+image->rerank", "input_modalities": [ "text", "image" ], "output_modalities": [ "rerank" ], "tokenizer": "Other", "instruct_type": null }, "description": "Llama Nemotron Rerank VL 1B V2 is a 1.7B multimodal reranking model from NVIDIA. It evaluates the relevance of document images and text against user queries, designed for vision RAG...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nvidia | nvidia/llama-nemotron-rerank-vl-1b-v2:free", "model_id": "nvidia/llama-nemotron-rerank-vl-1b-v2:free", "model_name": "NVIDIA: Llama Nemotron Rerank VL 1B V2", "context_length": 10240, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Nvidia", "tag": "nvidia", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "temperature", "max_tokens", "seed", "top_p" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "~anthropic/claude-fable-latest", "name": "Anthropic: Claude Fable Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "canonicalSlug": "~anthropic/claude-fable-latest", "contextLength": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "description": "This model always redirects to the latest model in the Claude Fable family.", "endpointCount": 0 } }, { "id": "anthropic/claude-fable-5@claude-platform-on-aws", "name": "Anthropic: Claude Fable 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-fable-5", "canonicalSlug": "anthropic/claude-5-fable-20260609", "servingProvider": "Claude Platform on AWS", "servingProviderSlug": "claude-platform-on-aws", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Claude Platform on AWS | anthropic/claude-5-fable-20260609", "model_id": "anthropic/claude-fable-5", "model_name": "Anthropic: Claude Fable 5", "context_length": 1000000, "pricing": { "prompt": "0.00001", "completion": "0.00005", "web_search": "0.01", "input_cache_read": "0.000001", "input_cache_write": "0.0000125", "input_cache_write_1h": "0.00002", "discount": 0 }, "provider_name": "Claude Platform on AWS", "tag": "claude-on-aws", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.90783410138249, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-fable-5@amazon-bedrock", "name": "Anthropic: Claude Fable 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-fable-5", "canonicalSlug": "anthropic/claude-5-fable-20260609", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-5-fable-20260609", "model_id": "anthropic/claude-fable-5", "model_name": "Anthropic: Claude Fable 5", "context_length": 1000000, "pricing": { "prompt": "0.00001", "completion": "0.00005", "web_search": "0.01", "input_cache_read": "0.000001", "input_cache_write": "0.0000125", "input_cache_write_1h": "0.00002", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-fable-5@azure", "name": "Anthropic: Claude Fable 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-fable-5", "canonicalSlug": "anthropic/claude-5-fable-20260609", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-5-fable-20260609", "model_id": "anthropic/claude-fable-5", "model_name": "Anthropic: Claude Fable 5", "context_length": 1000000, "pricing": { "prompt": "0.00001", "completion": "0.00005", "web_search": "0.01", "input_cache_read": "0.000001", "input_cache_write": "0.0000125", "input_cache_write_1h": "0.00002", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 97.60213143872114, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-fable-5@anthropic", "name": "Anthropic: Claude Fable 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-fable-5", "canonicalSlug": "anthropic/claude-5-fable-20260609", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...", "uptimeLast30m": 99.5, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-5-fable-20260609", "model_id": "anthropic/claude-fable-5", "model_name": "Anthropic: Claude Fable 5", "context_length": 1000000, "pricing": { "prompt": "0.00001", "completion": "0.00005", "web_search": "0.01", "input_cache_read": "0.000001", "input_cache_write": "0.0000125", "input_cache_write_1h": "0.00002", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.5, "uptime_last_5m": null, "uptime_last_1d": 99.86230742162996, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-fable-5@google", "name": "Anthropic: Claude Fable 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-fable-5", "canonicalSlug": "anthropic/claude-5-fable-20260609", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...", "uptimeLast30m": 99.97513056453619, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-5-fable-20260609", "model_id": "anthropic/claude-fable-5", "model_name": "Anthropic: Claude Fable 5", "context_length": 1000000, "pricing": { "prompt": "0.00001", "completion": "0.00005", "web_search": "0.01", "input_cache_read": "0.000001", "input_cache_write": "0.0000125", "input_cache_write_1h": "0.00002", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.97513056453619, "uptime_last_5m": 99.94410285075462, "uptime_last_1d": 99.9854759731098, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-fable-5@google", "name": "Anthropic: Claude Fable 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000055" } ], "cacheRead": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheWrite": [ { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-fable-5", "canonicalSlug": "anthropic/claude-5-fable-20260609", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-5-fable-20260609", "model_id": "anthropic/claude-fable-5", "model_name": "Anthropic: Claude Fable 5", "context_length": 1000000, "pricing": { "prompt": "0.000011", "completion": "0.000055", "web_search": "0.01", "input_cache_read": "0.0000011", "input_cache_write": "0.00001375", "input_cache_write_1h": "0.000022", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/europe", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-fable-5:batch@anthropic", "name": "Anthropic: Claude Fable 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-fable-5:batch", "canonicalSlug": "anthropic/claude-5-fable-20260609", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-5-fable-20260609:batch", "model_id": "anthropic/claude-fable-5:batch", "model_name": "Anthropic: Claude Fable 5", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nex-agi/nex-n2-pro@nex-agi", "name": "Nex AGI: Nex-N2-Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nex-agi/nex-n2-pro", "canonicalSlug": "nex-agi/nex-n2-pro", "servingProvider": "Nex AGI", "servingProviderSlug": "nex-agi", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "top_p", "top_k", "temperature", "max_tokens", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Nex-N2-Pro is an agentic mixture-of-experts model from Nex AGI, with 17B active parameters out of 397B total. Built on the Qwen3.5 architecture, it accepts text and image input and produces...", "uptimeLast30m": 99.50248756218906, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nex AGI | nex-agi/nex-n2-pro", "model_id": "nex-agi/nex-n2-pro", "model_name": "Nex AGI: Nex-N2-Pro", "context_length": 262144, "pricing": { "prompt": "0.00000025", "completion": "0.000001", "input_cache_read": "0.000000025", "discount": 0 }, "provider_name": "Nex AGI", "tag": "nex-agi/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "top_p", "top_k", "temperature", "max_tokens", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.50248756218906, "uptime_last_5m": 100, "uptime_last_1d": 98.91861761426979, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nex-agi/nex-n2-pro@siliconflow", "name": "Nex AGI: Nex-N2-Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nex-agi/nex-n2-pro", "canonicalSlug": "nex-agi/nex-n2-pro", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 256000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Nex-N2-Pro is an agentic mixture-of-experts model from Nex AGI, with 17B active parameters out of 397B total. Built on the Qwen3.5 architecture, it accepts text and image input and produces...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | nex-agi/nex-n2-pro", "model_id": "nex-agi/nex-n2-pro", "model_name": "Nex AGI: Nex-N2-Pro", "context_length": 262144, "pricing": { "prompt": "0.0000005", "completion": "0.0000025", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow", "quantization": "unknown", "max_completion_tokens": 256000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 98.3, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sourceful/riverflow-v2.5-pro@sourceful", "name": "Sourceful: Riverflow V2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sourceful/riverflow-v2.5-pro", "canonicalSlug": "sourceful/riverflow-v2.5-pro-20260605", "servingProvider": "Sourceful", "servingProviderSlug": "sourceful", "contextLength": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "reasoning_effort" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Riverflow V2.5 Pro is the most powerful variant of Sourceful's Riverflow 2.5 lineup, best for top-tier control and quality-sensitive outputs. The Riverflow 2.5 series is a unified text-to-image and image-to-image...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Sourceful | sourceful/riverflow-v2.5-pro-20260605", "model_id": "sourceful/riverflow-v2.5-pro", "model_name": "Sourceful: Riverflow V2.5 Pro", "context_length": 32768, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.0000311377245508982", "image_output": "0.0000311377245508982", "discount": 0 }, "provider_name": "Sourceful", "tag": "sourceful", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sourceful/riverflow-v2.5-fast@sourceful", "name": "Sourceful: Riverflow V2.5 Fast", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sourceful/riverflow-v2.5-fast", "canonicalSlug": "sourceful/riverflow-v2.5-fast-20260605", "servingProvider": "Sourceful", "servingProviderSlug": "sourceful", "contextLength": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "reasoning_effort" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Riverflow V2.5 Fast is the speed-optimized variant of Sourceful's Riverflow 2.5 lineup, best for production deployments and latency-critical workflows. The Riverflow 2.5 series is a unified text-to-image and image-to-image family...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Sourceful | sourceful/riverflow-v2.5-fast-20260605", "model_id": "sourceful/riverflow-v2.5-fast", "model_name": "Sourceful: Riverflow V2.5 Fast", "context_length": 32768, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.00000455089820359281", "image_output": "0.00000455089820359281", "discount": 0 }, "provider_name": "Sourceful", "tag": "sourceful", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3.5-content-safety:free@nvidia", "name": "NVIDIA: Nemotron 3.5 Content Safety", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3.5-content-safety:free", "canonicalSlug": "nvidia/nemotron-3.5-content-safety-20260604", "servingProvider": "Nvidia", "servingProviderSlug": "nvidia", "contextLength": 128000, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...", "uptimeLast30m": 95.6369982547993, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nvidia | nvidia/nemotron-3.5-content-safety-20260604:free", "model_id": "nvidia/nemotron-3.5-content-safety:free", "model_name": "NVIDIA: Nemotron 3.5 Content Safety", "context_length": 128000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Nvidia", "tag": "nvidia", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p" ], "status": 0, "uptime_last_30m": 95.6369982547993, "uptime_last_5m": 97.96954314720813, "uptime_last_1d": 98.05230346153333, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-ultra-550b-a55b@deepinfra", "name": "NVIDIA: Nemotron 3 Ultra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-ultra-550b-a55b", "canonicalSlug": "nvidia/nemotron-3-ultra-550b-a55b-20260604", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp4", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...", "uptimeLast30m": 86.49130628622382, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | nvidia/nemotron-3-ultra-550b-a55b-20260604", "model_id": "nvidia/nemotron-3-ultra-550b-a55b", "model_name": "NVIDIA: Nemotron 3 Ultra", "context_length": 262144, "pricing": { "prompt": "0.0000005", "completion": "0.0000022", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": -2, "uptime_last_30m": 86.49130628622382, "uptime_last_5m": 98.14814814814815, "uptime_last_1d": 91.81871975487772, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-ultra-550b-a55b@baseten", "name": "NVIDIA: Nemotron 3 Ultra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-ultra-550b-a55b", "canonicalSlug": "nvidia/nemotron-3-ultra-550b-a55b-20260604", "servingProvider": "BaseTen", "servingProviderSlug": "baseten", "contextLength": 202800, "maxCompletionTokens": 202800, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "BaseTen | nvidia/nemotron-3-ultra-550b-a55b-20260604", "model_id": "nvidia/nemotron-3-ultra-550b-a55b", "model_name": "NVIDIA: Nemotron 3 Ultra", "context_length": 202800, "pricing": { "prompt": "0.0000006", "completion": "0.0000024", "input_cache_read": "0.00000012", "discount": 0 }, "provider_name": "BaseTen", "tag": "baseten/fp4", "quantization": "fp4", "max_completion_tokens": 202800, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.91413564740809, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-ultra-550b-a55b@together", "name": "NVIDIA: Nemotron 3 Ultra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-ultra-550b-a55b", "canonicalSlug": "nvidia/nemotron-3-ultra-550b-a55b-20260604", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 512288, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tool_choice", "tools", "structured_outputs", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...", "uptimeLast30m": 99.1060291060291, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | nvidia/nemotron-3-ultra-550b-a55b-20260604", "model_id": "nvidia/nemotron-3-ultra-550b-a55b", "model_name": "NVIDIA: Nemotron 3 Ultra", "context_length": 512288, "pricing": { "prompt": "0.0000006", "completion": "0.0000036", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tool_choice", "tools", "structured_outputs", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.1060291060291, "uptime_last_5m": 100, "uptime_last_1d": 93.79090212022325, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-ultra-550b-a55b@venice", "name": "NVIDIA: Nemotron 3 Ultra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 3.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003125" } ], "cacheRead": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-ultra-550b-a55b", "canonicalSlug": "nvidia/nemotron-3-ultra-550b-a55b-20260604", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...", "uptimeLast30m": 96.47829647829647, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | nvidia/nemotron-3-ultra-550b-a55b-20260604", "model_id": "nvidia/nemotron-3-ultra-550b-a55b", "model_name": "NVIDIA: Nemotron 3 Ultra", "context_length": 256000, "pricing": { "prompt": "0.000000625", "completion": "0.000003125", "input_cache_read": "0.0000001875", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 96.47829647829647, "uptime_last_5m": 97.88732394366197, "uptime_last_1d": 88.31620741445715, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-ultra-550b-a55b:batch@together", "name": "NVIDIA: Nemotron 3 Ultra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-ultra-550b-a55b:batch", "canonicalSlug": "nvidia/nemotron-3-ultra-550b-a55b-20260604", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 512288, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tool_choice", "tools", "structured_outputs", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | nvidia/nemotron-3-ultra-550b-a55b-20260604:batch", "model_id": "nvidia/nemotron-3-ultra-550b-a55b:batch", "model_name": "NVIDIA: Nemotron 3 Ultra", "context_length": 512288, "pricing": { "prompt": "0.0000006", "completion": "0.0000036", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tool_choice", "tools", "structured_outputs", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-ultra-550b-a55b:free@nvidia", "name": "NVIDIA: Nemotron 3 Ultra", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-ultra-550b-a55b:free", "canonicalSlug": "nvidia/nemotron-3-ultra-550b-a55b-20260604", "servingProvider": "Nvidia", "servingProviderSlug": "nvidia", "contextLength": 1000000, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...", "uptimeLast30m": 97.48089835610095, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nvidia | nvidia/nemotron-3-ultra-550b-a55b-20260604:free", "model_id": "nvidia/nemotron-3-ultra-550b-a55b:free", "model_name": "NVIDIA: Nemotron 3 Ultra", "context_length": 1000000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Nvidia", "tag": "nvidia", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.48089835610095, "uptime_last_5m": 97.2493086886916, "uptime_last_1d": 99.67308918547928, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.7-plus@alibaba", "name": "Qwen: Qwen3.7 Plus", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000032" } ], "output": [ { "amount": 1.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000128" } ], "cacheRead": [ { "amount": 0.064, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000064" } ], "cacheWrite": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.7-plus", "canonicalSlug": "qwen/qwen3.7-plus-20260602", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "logprobs", "top_logprobs", "tools", "tool_choice", "structured_outputs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...", "uptimeLast30m": 99.99445491848729, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.7-plus-20260602", "model_id": "qwen/qwen3.7-plus", "model_name": "Qwen: Qwen3.7 Plus", "context_length": 1000000, "pricing": { "prompt": "0.00000032", "completion": "0.00000128", "input_cache_read": "0.000000064", "input_cache_write": "0.0000004", "discount": 0, "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.00000096", "completion": "0.00000384", "input_cache_read": "0.000000192", "input_cache_write": "0.0000012" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": 983616, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "logprobs", "top_logprobs", "tools", "tool_choice", "structured_outputs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 99.99445491848729, "uptime_last_5m": 100, "uptime_last_1d": 99.9911535431991, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "microsoft/mai-voice-2@azure", "name": "Microsoft: MAI-Voice-2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000022" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "microsoft/mai-voice-2", "canonicalSlug": "microsoft/mai-voice-2", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "max_completion_tokens" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "MAI-Voice-2 is an expressive text-to-speech model from Microsoft. It is suited for conversational assistants, media narration, accessibility, education, and other long-form voice applications. It supports 15 languages across 18 locales,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | microsoft/mai-voice-2", "model_id": "microsoft/mai-voice-2", "model_name": "Microsoft: MAI-Voice-2", "context_length": 0, "pricing": { "prompt": "0.000022", "completion": "0", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "max_completion_tokens" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "microsoft/mai-transcribe-1.5@azure", "name": "Microsoft: MAI-Transcribe 1.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 360000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.36" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "microsoft/mai-transcribe-1.5", "canonicalSlug": "microsoft/mai-transcribe-1.5", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "max_completion_tokens" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Other", "instruct_type": null }, "description": "MAI-Transcribe 1.5 is a multilingual speech-to-text model from Microsoft AI. It is suited for captions, call transcription, subtitling, accessibility, and other voice-enabled applications, with reliable transcription across 43 languages, diverse...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | microsoft/mai-transcribe-1.5", "model_id": "microsoft/mai-transcribe-1.5", "model_name": "Microsoft: MAI-Transcribe 1.5", "context_length": 0, "pricing": { "prompt": "0.36", "completion": "0", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "max_completion_tokens" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "microsoft/mai-image-2.5@azure", "name": "Microsoft: MAI-Image-2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "microsoft/mai-image-2.5", "canonicalSlug": "microsoft/mai-image-2.5", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 4096, "maxCompletionTokens": 1024, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "max_completion_tokens" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Microsoft's MAI-Image-2.5 is a high-quality image generation model available via Azure AI Foundry. It produces photorealistic and artistic images from text prompts with support for various aspect ratios.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | microsoft/mai-image-2.5", "model_id": "microsoft/mai-image-2.5", "model_name": "Microsoft: MAI-Image-2.5", "context_length": 4096, "pricing": { "prompt": "0.000005", "completion": "0", "image_token": "0.000047", "image_output": "0.000047", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 1024, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "max_completion_tokens" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@coreweave", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000023" } ], "output": [ { "amount": 0.96, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000096" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 99.61538461538461, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 262144, "pricing": { "prompt": "0.00000023", "completion": "0.00000096", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.61538461538461, "uptime_last_5m": 100, "uptime_last_1d": 99.89092340757857, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@gmicloud", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000024" } ], "output": [ { "amount": 0.96, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000096" } ], "cacheRead": [ { "amount": 0.048, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000048" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 97.8756884343037, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 1048576, "pricing": { "prompt": "0.00000024", "completion": "0.00000096", "input_cache_read": "0.000000048", "discount": 0.6 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": 97.8756884343037, "uptime_last_5m": 98.75, "uptime_last_1d": 99.03130740461748, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@deepinfra", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [ { "amount": 0.056, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000056" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 524288, "maxCompletionTokens": 512000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 98.5593220338983, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 524288, "pricing": { "prompt": "0.00000028", "completion": "0.0000011", "input_cache_read": "0.000000056", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 512000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 98.5593220338983, "uptime_last_5m": 99.23664122137404, "uptime_last_1d": 94.78841981896616, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@novita", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 1000000, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 99.51319248369195, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 1000000, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.51319248369195, "uptime_last_5m": 99.37952430196484, "uptime_last_1d": 99.53477770677563, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@venice", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 524288, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 524288, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.8753193042209, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@minimax", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 524288, "maxCompletionTokens": 512000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tool_choice", "tools" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 99.26905673512009, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 524288, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax/fp8", "quantization": "fp8", "max_completion_tokens": 512000, "max_prompt_tokens": 524288, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 99.26905673512009, "uptime_last_5m": 99.61538461538461, "uptime_last_1d": 99.33507357564085, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@atlascloud", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 524300, "maxCompletionTokens": 524288, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 99.23469387755102, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 524300, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 524288, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools" ], "status": 0, "uptime_last_30m": 99.23469387755102, "uptime_last_5m": 100, "uptime_last_1d": 99.56097054086351, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@together", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 524288, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 99.90485252140819, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 524288, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.90485252140819, "uptime_last_5m": 100, "uptime_last_1d": 98.37149120638216, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@streamlake", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 1000000, "maxCompletionTokens": 512000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 99.68304278922345, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 1000000, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0.5 }, "provider_name": "StreamLake", "tag": "streamlake/fp8", "quantization": "fp8", "max_completion_tokens": 512000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "stop" ], "status": 0, "uptime_last_30m": 99.68304278922345, "uptime_last_5m": 100, "uptime_last_1d": 98.99568868841429, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@parasail", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 1048576, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 98.48484848484848, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 1048576, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 98.48484848484848, "uptime_last_5m": 97.22222222222221, "uptime_last_1d": 98.65965475438387, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@morph", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "Morph", "servingProviderSlug": "morph", "contextLength": 256000, "maxCompletionTokens": 256000, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 99.53379953379954, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Morph | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 256000, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "discount": 0 }, "provider_name": "Morph", "tag": "morph/fp4", "quantization": "fp4", "max_completion_tokens": 256000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.53379953379954, "uptime_last_5m": 100, "uptime_last_1d": 94.38865465695669, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3@modelrun", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "ModelRun", "servingProviderSlug": "modelrun", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "presence_penalty", "repetition_penalty", "frequency_penalty", "top_p", "top_k", "stop", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": 99.69293756397134, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "ModelRun | minimax/minimax-m3-20260531", "model_id": "minimax/minimax-m3", "model_name": "MiniMax: MiniMax M3", "context_length": 1048576, "pricing": { "prompt": "0.00000075", "completion": "0.000003", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "ModelRun", "tag": "modelrun/fp4", "quantization": "fp4", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "presence_penalty", "repetition_penalty", "frequency_penalty", "top_p", "top_k", "stop", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.69293756397134, "uptime_last_5m": 100, "uptime_last_1d": 99.0540613865326, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m3:batch@together", "name": "MiniMax: MiniMax M3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m3:batch", "canonicalSlug": "minimax/minimax-m3-20260531", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 524288, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | minimax/minimax-m3-20260531:batch", "model_id": "minimax/minimax-m3:batch", "model_name": "MiniMax: MiniMax M3", "context_length": 524288, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "stepfun/step-3.7-flash@stepfun", "name": "StepFun: Step 3.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000115" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "stepfun/step-3.7-flash", "canonicalSlug": "stepfun/step-3.7-flash-20260528", "servingProvider": "StepFun", "servingProviderSlug": "stepfun", "contextLength": 256000, "maxCompletionTokens": 256000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "tools", "top_p", "stop", "frequency_penalty", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...", "uptimeLast30m": 95.10022271714922, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StepFun | stepfun/step-3.7-flash-20260528", "model_id": "stepfun/step-3.7-flash", "model_name": "StepFun: Step 3.7 Flash", "context_length": 256000, "pricing": { "prompt": "0.0000002", "completion": "0.00000115", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "StepFun", "tag": "stepfun/fp8", "quantization": "fp8", "max_completion_tokens": 256000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "tools", "top_p", "stop", "frequency_penalty", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 95.10022271714922, "uptime_last_5m": 95.91836734693877, "uptime_last_1d": 97.19094464857176, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "stepfun/step-3.7-flash@deepinfra", "name": "StepFun: Step 3.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000115" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "stepfun/step-3.7-flash", "canonicalSlug": "stepfun/step-3.7-flash-20260528", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | stepfun/step-3.7-flash-20260528", "model_id": "stepfun/step-3.7-flash", "model_name": "StepFun: Step 3.7 Flash", "context_length": 262144, "pricing": { "prompt": "0.0000002", "completion": "0.00000115", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.83210409546082, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "stepfun/step-3.7-flash@novita", "name": "StepFun: Step 3.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000115" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "stepfun/step-3.7-flash", "canonicalSlug": "stepfun/step-3.7-flash-20260528", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 256000, "quantization": "fp8", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...", "uptimeLast30m": 93.78109452736318, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | stepfun/step-3.7-flash-20260528", "model_id": "stepfun/step-3.7-flash", "model_name": "StepFun: Step 3.7 Flash", "context_length": 262144, "pricing": { "prompt": "0.0000002", "completion": "0.00000115", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 256000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "status": -2, "uptime_last_30m": 93.78109452736318, "uptime_last_5m": 94.5945945945946, "uptime_last_1d": 92.81555803294934, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8-fast@anthropic", "name": "Anthropic: Claude Opus 4.8 (Fast)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8-fast", "canonicalSlug": "anthropic/claude-4.8-opus-fast-20260528", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Fast-mode variant of [Opus 4.8](/anthropic/claude-opus-4.8) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 4.8.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.8-opus-fast-20260528", "model_id": "anthropic/claude-opus-4.8-fast", "model_name": "Anthropic: Claude Opus 4.8 (Fast)", "context_length": 1000000, "pricing": { "prompt": "0.00001", "completion": "0.00005", "web_search": "0.01", "input_cache_read": "0.000001", "input_cache_write": "0.0000125", "input_cache_write_1h": "0.00002", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8@amazon-bedrock", "name": "Anthropic: Claude Opus 4.8", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8", "canonicalSlug": "anthropic/claude-4.8-opus-20260528", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.8-opus-20260528", "model_id": "anthropic/claude-opus-4.8", "model_name": "Anthropic: Claude Opus 4.8", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.97968069666183, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8@azure", "name": "Anthropic: Claude Opus 4.8", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8", "canonicalSlug": "anthropic/claude-4.8-opus-20260528", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-4.8-opus-20260528", "model_id": "anthropic/claude-opus-4.8", "model_name": "Anthropic: Claude Opus 4.8", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Azure", "tag": "azure/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.24170616113744, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8@google", "name": "Anthropic: Claude Opus 4.8", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8", "canonicalSlug": "anthropic/claude-4.8-opus-20260528", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "uptimeLast30m": 99.9794597925439, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.8-opus-20260528", "model_id": "anthropic/claude-opus-4.8", "model_name": "Anthropic: Claude Opus 4.8", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.9794597925439, "uptime_last_5m": 100, "uptime_last_1d": 99.98411458678069, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8@anthropic", "name": "Anthropic: Claude Opus 4.8", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8", "canonicalSlug": "anthropic/claude-4.8-opus-20260528", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "uptimeLast30m": 98.21428571428571, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.8-opus-20260528", "model_id": "anthropic/claude-opus-4.8", "model_name": "Anthropic: Claude Opus 4.8", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.21428571428571, "uptime_last_5m": 97.36842105263158, "uptime_last_1d": 99.72407812530794, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8@azure", "name": "Anthropic: Claude Opus 4.8", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8", "canonicalSlug": "anthropic/claude-4.8-opus-20260528", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-4.8-opus-20260528", "model_id": "anthropic/claude-opus-4.8", "model_name": "Anthropic: Claude Opus 4.8", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8@claude-platform-on-aws", "name": "Anthropic: Claude Opus 4.8", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8", "canonicalSlug": "anthropic/claude-4.8-opus-20260528", "servingProvider": "Claude Platform on AWS", "servingProviderSlug": "claude-platform-on-aws", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "uptimeLast30m": 99.93032168157009, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Claude Platform on AWS | anthropic/claude-4.8-opus-20260528", "model_id": "anthropic/claude-opus-4.8", "model_name": "Anthropic: Claude Opus 4.8", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Claude Platform on AWS", "tag": "claude-on-aws", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.93032168157009, "uptime_last_5m": 100, "uptime_last_1d": 99.87704211743156, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8@google", "name": "Anthropic: Claude Opus 4.8", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8", "canonicalSlug": "anthropic/claude-4.8-opus-20260528", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.8-opus-20260528", "model_id": "anthropic/claude-opus-4.8", "model_name": "Anthropic: Claude Opus 4.8", "context_length": 1000000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/europe", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8@amazon-bedrock", "name": "Anthropic: Claude Opus 4.8", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8", "canonicalSlug": "anthropic/claude-4.8-opus-20260528", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.8-opus-20260528", "model_id": "anthropic/claude-opus-4.8", "model_name": "Anthropic: Claude Opus 4.8", "context_length": 1000000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8@google", "name": "Anthropic: Claude Opus 4.8", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8", "canonicalSlug": "anthropic/claude-4.8-opus-20260528", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.8-opus-20260528", "model_id": "anthropic/claude-opus-4.8", "model_name": "Anthropic: Claude Opus 4.8", "context_length": 1000000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8@amazon-bedrock", "name": "Anthropic: Claude Opus 4.8", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8", "canonicalSlug": "anthropic/claude-4.8-opus-20260528", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.8-opus-20260528", "model_id": "anthropic/claude-opus-4.8", "model_name": "Anthropic: Claude Opus 4.8", "context_length": 1000000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.8:batch@anthropic", "name": "Anthropic: Claude Opus 4.8", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [ { "amount": 3.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.8:batch", "canonicalSlug": "anthropic/claude-4.8-opus-20260528", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.8-opus-20260528:batch", "model_id": "anthropic/claude-opus-4.8:batch", "model_name": "Anthropic: Claude Opus 4.8", "context_length": 1000000, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "input_cache_write_1h": "0.000005", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/parakeet-tdt-0.6b-v3@together", "name": "NVIDIA: Parakeet TDT 0.6B v3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1500, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/parakeet-tdt-0.6b-v3", "canonicalSlug": "nvidia/parakeet-tdt-0.6b-v3", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Other", "instruct_type": null }, "description": "Parakeet TDT 0.6B v3 is NVIDIA's 600M-parameter multilingual speech-to-text model built on the FastConformer-TDT architecture. Trained on the Granary dataset (670,000+ hours of audio), it supports automatic language detection across...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | nvidia/parakeet-tdt-0.6b-v3", "model_id": "nvidia/parakeet-tdt-0.6b-v3", "model_name": "NVIDIA: Parakeet TDT 0.6B v3", "context_length": 0, "pricing": { "prompt": "0.0015", "completion": "0", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.7-max@alibaba", "name": "Qwen: Qwen3.7 Max", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.475, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001475" } ], "output": [ { "amount": 4.425, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004425" } ], "cacheRead": [ { "amount": 0.295, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000295" } ], "cacheWrite": [ { "amount": 1.84375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000184375" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.7-max", "canonicalSlug": "qwen/qwen3.7-max-20260520", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.7-max-20260520", "model_id": "qwen/qwen3.7-max", "model_name": "Qwen: Qwen3.7 Max", "context_length": 1000000, "pricing": { "prompt": "0.000001475", "completion": "0.000004425", "input_cache_read": "0.000000295", "input_cache_write": "0.00000184375", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": 983616, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99944419124269, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-build-0.1@xai", "name": "SpaceXAI: Grok Build 0.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-build-0.1", "canonicalSlug": "x-ai/grok-build-0.1-20260520", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 256000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-build-0.1-20260520", "model_id": "x-ai/grok-build-0.1", "model_name": "SpaceXAI: Grok Build 0.1", "context_length": 256000, "pricing": { "prompt": "0.000001", "completion": "0.000002", "web_search": "0.005", "input_cache_read": "0.0000002", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000002", "completion": "0.000004", "input_cache_read": "0.0000004" } ] }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.98326639892905, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-build-0.1@xai", "name": "SpaceXAI: Grok Build 0.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-build-0.1", "canonicalSlug": "x-ai/grok-build-0.1-20260520", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 256000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-build-0.1-20260520", "model_id": "x-ai/grok-build-0.1", "model_name": "SpaceXAI: Grok Build 0.1", "context_length": 256000, "pricing": { "prompt": "0.000002", "completion": "0.000004", "web_search": "0.005", "input_cache_read": "0.0000004", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000008", "input_cache_read": "0.0000008" } ] }, "provider_name": "xAI", "tag": "xai/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.98326639892905, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-build-0.1@xai", "name": "SpaceXAI: Grok Build 0.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-build-0.1", "canonicalSlug": "x-ai/grok-build-0.1-20260520", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 256000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-build-0.1-20260520", "model_id": "x-ai/grok-build-0.1", "model_name": "SpaceXAI: Grok Build 0.1", "context_length": 256000, "pricing": { "prompt": "0.000001", "completion": "0.000002", "web_search": "0.005", "input_cache_read": "0.0000002", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000002", "completion": "0.000004", "input_cache_read": "0.0000004" } ] }, "provider_name": "xAI", "tag": "xai/zdr", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-build-0.1@xai", "name": "SpaceXAI: Grok Build 0.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-build-0.1", "canonicalSlug": "x-ai/grok-build-0.1-20260520", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 256000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-build-0.1-20260520", "model_id": "x-ai/grok-build-0.1", "model_name": "SpaceXAI: Grok Build 0.1", "context_length": 256000, "pricing": { "prompt": "0.000002", "completion": "0.000004", "web_search": "0.005", "input_cache_read": "0.0000004", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000008", "input_cache_read": "0.0000008" } ] }, "provider_name": "xAI", "tag": "xai/zdr/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-embedding-2@google-ai-studio", "name": "Google: Gemini Embedding 2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.0000002" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-embedding-2", "canonicalSlug": "google/gemini-embedding-2", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text+image+file+audio+video->embeddings", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "embeddings" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini Embedding 2 is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It supports...", "uptimeLast30m": 99.98783602968008, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-embedding-2", "model_id": "google/gemini-embedding-2", "model_name": "Google: Gemini Embedding 2", "context_length": 8192, "pricing": { "prompt": "0.0000002", "completion": "0", "image": "0.00000045", "audio": "0.0000065", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": 99.98783602968008, "uptime_last_5m": 100, "uptime_last_1d": 99.9923488654494, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-embedding-2@google", "name": "Google: Gemini Embedding 2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.0000002" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-embedding-2", "canonicalSlug": "google/gemini-embedding-2", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text+image+file+audio+video->embeddings", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "embeddings" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini Embedding 2 is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It supports...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-embedding-2", "model_id": "google/gemini-embedding-2", "model_name": "Google: Gemini Embedding 2", "context_length": 8192, "pricing": { "prompt": "0.0000002", "completion": "0", "image": "0.00000045", "audio": "0.0000065", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99974934956212, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-embedding-2@google", "name": "Google: Gemini Embedding 2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.0000002" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-embedding-2", "canonicalSlug": "google/gemini-embedding-2", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text+image+file+audio+video->embeddings", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "embeddings" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini Embedding 2 is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It supports...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-embedding-2", "model_id": "google/gemini-embedding-2", "model_name": "Google: Gemini Embedding 2", "context_length": 8192, "pricing": { "prompt": "0.0000002", "completion": "0", "image": "0.00000045", "audio": "0.0000065", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99897399855847, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash@google", "name": "Google: Gemini 3.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 0.0000015, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000015" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash", "canonicalSlug": "google/gemini-3.5-flash-20260519", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...", "uptimeLast30m": 92.73439746908254, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.5-flash-20260519", "model_id": "google/gemini-3.5-flash", "model_name": "Google: Gemini 3.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.0000015", "completion": "0.000009", "image": "0.0000015", "audio": "0.000003", "input_audio_cache": "0.0000003", "web_search": "0.014", "internal_reasoning": "0.000009", "input_cache_read": "0.00000015", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": -2, "uptime_last_30m": 92.73439746908254, "uptime_last_5m": 99.87681694998768, "uptime_last_1d": 99.39816194583153, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash@google", "name": "Google: Gemini 3.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 7.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000075" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash", "canonicalSlug": "google/gemini-3.5-flash-20260519", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...", "uptimeLast30m": 92.73439746908254, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.5-flash-20260519", "model_id": "google/gemini-3.5-flash", "model_name": "Google: Gemini 3.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000075", "completion": "0.0000045", "image": "0.00000075", "audio": "0.0000015", "input_audio_cache": "0.00000015", "web_search": "0.014", "internal_reasoning": "0.0000045", "input_cache_read": "0.000000075", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": -2, "uptime_last_30m": 92.73439746908254, "uptime_last_5m": 99.87681694998768, "uptime_last_1d": 99.39816194583153, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash@google", "name": "Google: Gemini 3.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "output": [ { "amount": 16.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000162" } ], "cacheRead": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 0.0000027, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000027" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 16.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000162" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash", "canonicalSlug": "google/gemini-3.5-flash-20260519", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...", "uptimeLast30m": 92.73439746908254, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.5-flash-20260519", "model_id": "google/gemini-3.5-flash", "model_name": "Google: Gemini 3.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.0000027", "completion": "0.0000162", "image": "0.0000027", "audio": "0.0000054", "input_audio_cache": "0.00000054", "web_search": "0.014", "internal_reasoning": "0.0000162", "input_cache_read": "0.00000027", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": -2, "uptime_last_30m": 92.73439746908254, "uptime_last_5m": 99.87681694998768, "uptime_last_1d": 99.39816194583153, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash@google-ai-studio", "name": "Google: Gemini 3.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 0.0000015, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000015" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash", "canonicalSlug": "google/gemini-3.5-flash-20260519", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...", "uptimeLast30m": 99.83664134607531, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.5-flash-20260519", "model_id": "google/gemini-3.5-flash", "model_name": "Google: Gemini 3.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.0000015", "completion": "0.000009", "image": "0.0000015", "audio": "0.000003", "input_audio_cache": "0.0000003", "web_search": "0.014", "internal_reasoning": "0.000009", "input_cache_read": "0.00000015", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.83664134607531, "uptime_last_5m": 99.74811083123426, "uptime_last_1d": 99.75103734439834, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash@google-ai-studio", "name": "Google: Gemini 3.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 7.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000075" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash", "canonicalSlug": "google/gemini-3.5-flash-20260519", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...", "uptimeLast30m": 99.83664134607531, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.5-flash-20260519", "model_id": "google/gemini-3.5-flash", "model_name": "Google: Gemini 3.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000075", "completion": "0.0000045", "image": "0.00000075", "audio": "0.0000015", "input_audio_cache": "0.00000015", "web_search": "0.014", "internal_reasoning": "0.0000045", "input_cache_read": "0.000000075", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.83664134607531, "uptime_last_5m": 99.74811083123426, "uptime_last_1d": 99.75103734439834, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash@google-ai-studio", "name": "Google: Gemini 3.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "output": [ { "amount": 16.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000162" } ], "cacheRead": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 0.0000027, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000027" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 16.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000162" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash", "canonicalSlug": "google/gemini-3.5-flash-20260519", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...", "uptimeLast30m": 99.83664134607531, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.5-flash-20260519", "model_id": "google/gemini-3.5-flash", "model_name": "Google: Gemini 3.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.0000027", "completion": "0.0000162", "image": "0.0000027", "audio": "0.0000054", "input_audio_cache": "0.00000054", "web_search": "0.014", "internal_reasoning": "0.0000162", "input_cache_read": "0.00000027", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.83664134607531, "uptime_last_5m": 99.74811083123426, "uptime_last_1d": 99.75103734439834, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash@google", "name": "Google: Gemini 3.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.6500000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000165" } ], "output": [ { "amount": 9.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000099" } ], "cacheRead": [ { "amount": 0.165, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000165" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 0.00000165, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000165" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 9.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000099" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash", "canonicalSlug": "google/gemini-3.5-flash-20260519", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.5-flash-20260519", "model_id": "google/gemini-3.5-flash", "model_name": "Google: Gemini 3.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000165", "completion": "0.0000099", "image": "0.00000165", "audio": "0.0000033", "input_audio_cache": "0.00000033", "web_search": "0.014", "internal_reasoning": "0.0000099", "input_cache_read": "0.000000165", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.5-flash:batch@google", "name": "Google: Gemini 3.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [ { "amount": 7.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000075" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.5-flash:batch", "canonicalSlug": "google/gemini-3.5-flash-20260519", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.5-flash-20260519:batch", "model_id": "google/gemini-3.5-flash:batch", "model_name": "Google: Gemini 3.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000075", "completion": "0.0000045", "image": "0.00000075", "audio": "0.0000015", "input_audio_cache": "0.00000015", "web_search": "0.014", "internal_reasoning": "0.0000045", "input_cache_read": "0.000000075", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-imagine-video@xai", "name": "SpaceXAI: Grok Imagine Video", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-imagine-video", "canonicalSlug": "x-ai/grok-imagine-video-20260512", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok Imagine Video is SpaceXAI's fast, text-, image-, and reference-conditioned video generation model. It produces short videos (1–15 seconds, 24 fps) at 480p or 720p across seven aspect ratios -...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-imagine-video-20260512", "model_id": "x-ai/grok-imagine-video", "model_name": "SpaceXAI: Grok Imagine Video", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-imagine-image-quality@xai", "name": "SpaceXAI: Grok Imagine Image Quality", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-imagine-image-quality", "canonicalSlug": "x-ai/grok-imagine-image-quality-20260512", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok Imagine Image Quality is SpaceXAI's fast, high-fidelity image generation and editing model. It accepts text prompts and optional reference images, producing photorealistic outputs at 1K or 2K across a...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-imagine-image-quality-20260512", "model_id": "x-ai/grok-imagine-image-quality", "model_name": "SpaceXAI: Grok Imagine Image Quality", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image": "0.01", "image_token": "0.0000119760479041916", "image_output": "0.0000119760479041916", "discount": 0 }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/voxtral-mini-transcribe@mistral", "name": "Mistral: Voxtral Mini Transcribe", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.003" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/voxtral-mini-transcribe", "canonicalSlug": "mistralai/voxtral-mini-transcribe-2602", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Voxtral Mini Transcribe is Mistral's speech-to-text model, derived from the Voxtral Mini family. It accepts audio input and returns transcribed text via the standard transcription API. Suited for transcribing meetings,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/voxtral-mini-transcribe-2602", "model_id": "mistralai/voxtral-mini-transcribe", "model_name": "Mistral: Voxtral Mini Transcribe", "context_length": 0, "pricing": { "prompt": "0.003", "completion": "0", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-voice-tts-1.0@xai", "name": "SpaceXAI: Grok Voice TTS 1.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-voice-tts-1.0", "canonicalSlug": "x-ai/grok-voice-tts-1.0", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 15000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok Voice TTS 1.0 is a text-to-speech model from SpaceXAI. It converts text into spoken audio across 20+ languages with automatic language detection, and offers five built-in voices (Eve, Ara,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-voice-tts-1.0", "model_id": "x-ai/grok-voice-tts-1.0", "model_name": "SpaceXAI: Grok Voice TTS 1.0", "context_length": 15000, "pricing": { "prompt": "0.000015", "completion": "0", "discount": 0 }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-asr-flash-2026-02-10@alibaba", "name": "Qwen: Qwen3 ASR Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000035" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-asr-flash-2026-02-10", "canonicalSlug": "qwen/qwen3-asr-flash-2026-02-10", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-ASR-Flash is Alibaba's automatic speech recognition service, built on the Qwen3-Omni foundation and trained on tens of millions of hours of multimodal speech data. The model handles 11 languages —...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-asr-flash-2026-02-10", "model_id": "qwen/qwen3-asr-flash-2026-02-10", "model_name": "Qwen: Qwen3 ASR Flash", "context_length": 0, "pricing": { "prompt": "0.000035", "completion": "0", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "recraft/recraft-v4.1-pro-vector@recraft", "name": "Recraft: Recraft V4.1 Pro Vector", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "recraft/recraft-v4.1-pro-vector", "canonicalSlug": "recraft/recraft-v4.1-pro-vector-20260514", "servingProvider": "Recraft", "servingProviderSlug": "recraft", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Recraft V4.1 Pro Vector is the vector (SVG) variant of Recraft V4.1 Pro, tuned for high aesthetics. It supports text and image inputs and produces higher-resolution SVG image output across...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Recraft | recraft/recraft-v4.1-pro-vector-20260514", "model_id": "recraft/recraft-v4.1-pro-vector", "model_name": "Recraft: Recraft V4.1 Pro Vector", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.0000718562874251497", "image_output": "0.0000718562874251497", "discount": 0 }, "provider_name": "Recraft", "tag": "recraft", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "recraft/recraft-v4.1-vector@recraft", "name": "Recraft: Recraft V4.1 Vector", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "recraft/recraft-v4.1-vector", "canonicalSlug": "recraft/recraft-v4.1-vector-20260514", "servingProvider": "Recraft", "servingProviderSlug": "recraft", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Recraft V4.1 Vector is the vector (SVG) variant of Recraft V4.1, tuned for high aesthetics. It supports text and image inputs and produces SVG image output across multiple aspect ratios,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Recraft | recraft/recraft-v4.1-vector-20260514", "model_id": "recraft/recraft-v4.1-vector", "model_name": "Recraft: Recraft V4.1 Vector", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.0000191616766467066", "image_output": "0.0000191616766467066", "discount": 0 }, "provider_name": "Recraft", "tag": "recraft", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "recraft/recraft-v4.1-utility-pro@recraft", "name": "Recraft: Recraft V4.1 Utility Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "recraft/recraft-v4.1-utility-pro", "canonicalSlug": "recraft/recraft-v4.1-utility-pro-20260514", "servingProvider": "Recraft", "servingProviderSlug": "recraft", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Recraft V4.1 Utility Pro is a general-purpose image generation model from Recraft. It supports text and image inputs with image output at ~2K resolution across multiple aspect ratios — double...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Recraft | recraft/recraft-v4.1-utility-pro-20260514", "model_id": "recraft/recraft-v4.1-utility-pro", "model_name": "Recraft: Recraft V4.1 Utility Pro", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.0000502994011976048", "image_output": "0.0000502994011976048", "discount": 0 }, "provider_name": "Recraft", "tag": "recraft", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "recraft/recraft-v4.1-utility@recraft", "name": "Recraft: Recraft V4.1 Utility", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "recraft/recraft-v4.1-utility", "canonicalSlug": "recraft/recraft-v4.1-utility-20260514", "servingProvider": "Recraft", "servingProviderSlug": "recraft", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Recraft V4.1 Utility is a general-purpose image generation model from Recraft. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios, with typical generation...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Recraft | recraft/recraft-v4.1-utility-20260514", "model_id": "recraft/recraft-v4.1-utility", "model_name": "Recraft: Recraft V4.1 Utility", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.00000838323353293413", "image_output": "0.00000838323353293413", "discount": 0 }, "provider_name": "Recraft", "tag": "recraft", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "recraft/recraft-v4.1-pro@recraft", "name": "Recraft: Recraft V4.1 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "recraft/recraft-v4.1-pro", "canonicalSlug": "recraft/recraft-v4.1-pro-20260514", "servingProvider": "Recraft", "servingProviderSlug": "recraft", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Recraft V4.1 Pro is an image generation model from Recraft tuned for high aesthetics. It supports text and image inputs with image output at ~2K resolution across multiple aspect ratios...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Recraft | recraft/recraft-v4.1-pro-20260514", "model_id": "recraft/recraft-v4.1-pro", "model_name": "Recraft: Recraft V4.1 Pro", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.0000502994011976048", "image_output": "0.0000502994011976048", "discount": 0 }, "provider_name": "Recraft", "tag": "recraft", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "recraft/recraft-v4.1@recraft", "name": "Recraft: Recraft V4.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "recraft/recraft-v4.1", "canonicalSlug": "recraft/recraft-v4.1-20260514", "servingProvider": "Recraft", "servingProviderSlug": "recraft", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Recraft V4.1 is an image generation model from Recraft tuned for high aesthetics. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios, with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Recraft | recraft/recraft-v4.1-20260514", "model_id": "recraft/recraft-v4.1", "model_name": "Recraft: Recraft V4.1", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.00000838323353293413", "image_output": "0.00000838323353293413", "discount": 0 }, "provider_name": "Recraft", "tag": "recraft", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "recraft/recraft-v4-pro-vector@recraft", "name": "Recraft: Recraft V4 Pro Vector", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "recraft/recraft-v4-pro-vector", "canonicalSlug": "recraft/recraft-v4-pro-vector-20260514", "servingProvider": "Recraft", "servingProviderSlug": "recraft", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Recraft V4 Pro Vector is the vector (SVG) variant of Recraft V4 Pro. It supports text and image inputs and produces vector image output across multiple aspect ratios at the...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Recraft | recraft/recraft-v4-pro-vector-20260514", "model_id": "recraft/recraft-v4-pro-vector", "model_name": "Recraft: Recraft V4 Pro Vector", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.0000718562874251497", "image_output": "0.0000718562874251497", "discount": 0 }, "provider_name": "Recraft", "tag": "recraft", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "recraft/recraft-v4-vector@recraft", "name": "Recraft: Recraft V4 Vector", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "recraft/recraft-v4-vector", "canonicalSlug": "recraft/recraft-v4-vector-20260514", "servingProvider": "Recraft", "servingProviderSlug": "recraft", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Recraft V4 Vector is the vector (SVG) variant of Recraft V4. It supports text and image inputs and produces vector image output across multiple aspect ratios. Compared to the raster...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Recraft | recraft/recraft-v4-vector-20260514", "model_id": "recraft/recraft-v4-vector", "model_name": "Recraft: Recraft V4 Vector", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.0000191616766467066", "image_output": "0.0000191616766467066", "discount": 0 }, "provider_name": "Recraft", "tag": "recraft", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.7-fast@anthropic", "name": "Anthropic: Claude Opus 4.7 (Fast)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 150, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00015" } ], "cacheRead": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheWrite": [ { "amount": 37.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.7-fast", "canonicalSlug": "anthropic/claude-4.7-opus-fast-20260512", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Fast-mode variant of [Opus 4.7](/anthropic/claude-opus-4.7) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.7-opus-fast-20260512", "model_id": "anthropic/claude-opus-4.7-fast", "model_name": "Anthropic: Claude Opus 4.7 (Fast)", "context_length": 1000000, "pricing": { "prompt": "0.00003", "completion": "0.00015", "web_search": "0.01", "input_cache_read": "0.000003", "input_cache_write": "0.0000375", "input_cache_write_1h": "0.00006", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "perceptron/perceptron-mk1@perceptron", "name": "Perceptron: Perceptron Mk1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "perceptron/perceptron-mk1", "canonicalSlug": "perceptron/perceptron-mk1-20260512", "servingProvider": "Perceptron", "servingProviderSlug": "perceptron", "contextLength": 32768, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Perceptron | perceptron/perceptron-mk1-20260512", "model_id": "perceptron/perceptron-mk1", "model_name": "Perceptron: Perceptron Mk1", "context_length": 32768, "pricing": { "prompt": "0.00000015", "completion": "0.0000015", "discount": 0 }, "provider_name": "Perceptron", "tag": "perceptron", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "inclusionai/ring-2.6-1t@novita", "name": "inclusionAI: Ring-2.6-1T", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "output": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "inclusionai/ring-2.6-1t", "canonicalSlug": "inclusionai/ring-2.6-1t-20260508", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Ring-2.6-1T is a 1T-parameter-scale thinking model with 63B active parameters, built for real-world agent workflows that require both strong capability and operational efficiency. It is optimized for coding agents, tool...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | inclusionai/ring-2.6-1t-20260508", "model_id": "inclusionai/ring-2.6-1t", "model_name": "inclusionAI: Ring-2.6-1T", "context_length": 262144, "pricing": { "prompt": "0.000000075", "completion": "0.000000625", "input_cache_read": "0.000000015", "discount": 0.75 }, "provider_name": "Novita", "tag": "novita", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "recraft/recraft-v4-pro@recraft", "name": "Recraft: Recraft V4 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "recraft/recraft-v4-pro", "canonicalSlug": "recraft/recraft-v4-pro-20260413", "servingProvider": "Recraft", "servingProviderSlug": "recraft", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Recraft V4 Pro is an image generation model from Recraft. It supports text and image inputs with image output at ~2K resolution across multiple aspect ratios, double the resolution of...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Recraft | recraft/recraft-v4-pro-20260413", "model_id": "recraft/recraft-v4-pro", "model_name": "Recraft: Recraft V4 Pro", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.0000598802395209581", "image_output": "0.0000598802395209581", "discount": 0 }, "provider_name": "Recraft", "tag": "recraft", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "recraft/recraft-v4@recraft", "name": "Recraft: Recraft V4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "recraft/recraft-v4", "canonicalSlug": "recraft/recraft-v4-20260413", "servingProvider": "Recraft", "servingProviderSlug": "recraft", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Recraft V4 is an image generation model from Recraft. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios. It delivers stronger compositional judgment,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Recraft | recraft/recraft-v4-20260413", "model_id": "recraft/recraft-v4", "model_name": "Recraft: Recraft V4", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.00000958083832335329", "image_output": "0.00000958083832335329", "discount": 0 }, "provider_name": "Recraft", "tag": "recraft", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "recraft/recraft-v3@recraft", "name": "Recraft: Recraft V3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "recraft/recraft-v3", "canonicalSlug": "recraft/recraft-v3-20260413", "servingProvider": "Recraft", "servingProviderSlug": "recraft", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Recraft V3 is an image generation model from Recraft. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios. Supports the following `image_config` parameters:...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Recraft | recraft/recraft-v3-20260413", "model_id": "recraft/recraft-v3", "model_name": "Recraft: Recraft V3", "context_length": 65536, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.00000958083832335329", "image_output": "0.00000958083832335329", "discount": 0 }, "provider_name": "Recraft", "tag": "recraft", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite@google", "name": "Google: Gemini 3.1 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 2.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000025" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite", "canonicalSlug": "google/gemini-3.1-flash-lite-20260507", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...", "uptimeLast30m": 98.34880707552216, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-flash-lite-20260507", "model_id": "google/gemini-3.1-flash-lite", "model_name": "Google: Gemini 3.1 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image": "0.00000025", "audio": "0.0000005", "input_audio_cache": "0.00000005", "web_search": "0.014", "internal_reasoning": "0.0000015", "input_cache_read": "0.000000025", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.34880707552216, "uptime_last_5m": 98.10744292624967, "uptime_last_1d": 97.96597766244338, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite@google", "name": "Google: Gemini 3.1 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [ { "amount": 0.012499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000125" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 1.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite", "canonicalSlug": "google/gemini-3.1-flash-lite-20260507", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...", "uptimeLast30m": 98.34880707552216, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-flash-lite-20260507", "model_id": "google/gemini-3.1-flash-lite", "model_name": "Google: Gemini 3.1 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.000000125", "completion": "0.00000075", "image": "0.000000125", "audio": "0.00000025", "input_audio_cache": "0.000000025", "web_search": "0.014", "internal_reasoning": "0.00000075", "input_cache_read": "0.0000000125", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.34880707552216, "uptime_last_5m": 98.10744292624967, "uptime_last_1d": 97.96597766244338, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite@google", "name": "Google: Gemini 3.1 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "cacheRead": [ { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000045" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 4.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000045" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite", "canonicalSlug": "google/gemini-3.1-flash-lite-20260507", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...", "uptimeLast30m": 98.34880707552216, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-flash-lite-20260507", "model_id": "google/gemini-3.1-flash-lite", "model_name": "Google: Gemini 3.1 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000045", "completion": "0.0000027", "image": "0.00000045", "audio": "0.0000009", "input_audio_cache": "0.00000009", "web_search": "0.014", "internal_reasoning": "0.0000027", "input_cache_read": "0.000000045", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.34880707552216, "uptime_last_5m": 98.10744292624967, "uptime_last_1d": 97.96597766244338, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite@google-ai-studio", "name": "Google: Gemini 3.1 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 2.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000025" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite", "canonicalSlug": "google/gemini-3.1-flash-lite-20260507", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...", "uptimeLast30m": 99.64454729293055, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-flash-lite-20260507", "model_id": "google/gemini-3.1-flash-lite", "model_name": "Google: Gemini 3.1 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image": "0.00000025", "audio": "0.0000005", "input_audio_cache": "0.00000005", "web_search": "0.014", "internal_reasoning": "0.0000015", "input_cache_read": "0.000000025", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.64454729293055, "uptime_last_5m": 99.73491846734677, "uptime_last_1d": 99.69038123084115, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite@google-ai-studio", "name": "Google: Gemini 3.1 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [ { "amount": 0.012499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000125" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 1.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite", "canonicalSlug": "google/gemini-3.1-flash-lite-20260507", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...", "uptimeLast30m": 99.64454729293055, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-flash-lite-20260507", "model_id": "google/gemini-3.1-flash-lite", "model_name": "Google: Gemini 3.1 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.000000125", "completion": "0.00000075", "image": "0.000000125", "audio": "0.00000025", "input_audio_cache": "0.000000025", "web_search": "0.014", "internal_reasoning": "0.00000075", "input_cache_read": "0.0000000125", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.64454729293055, "uptime_last_5m": 99.73491846734677, "uptime_last_1d": 99.69038123084115, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite@google-ai-studio", "name": "Google: Gemini 3.1 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "cacheRead": [ { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000045" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 4.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000045" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite", "canonicalSlug": "google/gemini-3.1-flash-lite-20260507", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...", "uptimeLast30m": 99.64454729293055, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-flash-lite-20260507", "model_id": "google/gemini-3.1-flash-lite", "model_name": "Google: Gemini 3.1 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000045", "completion": "0.0000027", "image": "0.00000045", "audio": "0.0000009", "input_audio_cache": "0.00000009", "web_search": "0.014", "internal_reasoning": "0.0000027", "input_cache_read": "0.000000045", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.64454729293055, "uptime_last_5m": 99.73491846734677, "uptime_last_1d": 99.69038123084115, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite@google", "name": "Google: Gemini 3.1 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "output": [ { "amount": 1.6500000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000165" } ], "cacheRead": [ { "amount": 0.0275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000275" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 2.75e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000275" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.6500000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000165" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite", "canonicalSlug": "google/gemini-3.1-flash-lite-20260507", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-flash-lite-20260507", "model_id": "google/gemini-3.1-flash-lite", "model_name": "Google: Gemini 3.1 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.000000275", "completion": "0.00000165", "image": "0.000000275", "audio": "0.00000055", "input_audio_cache": "0.000000055", "web_search": "0.014", "internal_reasoning": "0.00000165", "input_cache_read": "0.0000000275", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite:batch@google", "name": "Google: Gemini 3.1 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [ { "amount": 0.012499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000125" } ], "cacheWrite": [], "other": [ { "amount": 1.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite:batch", "canonicalSlug": "google/gemini-3.1-flash-lite-20260507", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-flash-lite-20260507:batch", "model_id": "google/gemini-3.1-flash-lite:batch", "model_name": "Google: Gemini 3.1 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.000000125", "completion": "0.00000075", "image": "0.000000125", "audio": "0.00000025", "input_audio_cache": "0.000000025", "web_search": "0.014", "internal_reasoning": "0.00000075", "input_cache_read": "0.0000000125", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-chat-latest@openai", "name": "OpenAI: GPT Chat Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-chat-latest", "canonicalSlug": "openai/gpt-chat-latest-20260505", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-chat-latest-20260505", "model_id": "openai/gpt-chat-latest", "model_name": "OpenAI: GPT Chat Latest", "context_length": 400000, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/chirp-3@google", "name": "Google: Chirp 3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 16000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.016" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/chirp-3", "canonicalSlug": "google/chirp-3", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "Other", "instruct_type": null }, "description": "Chirp 3 is Google's latest multilingual speech-to-text model. It offers enhanced transcription accuracy across 24 GA languages and 77+ preview languages, with support for automatic language detection, automatic punctuation, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/chirp-3", "model_id": "google/chirp-3", "model_name": "Google: Chirp 3", "context_length": 0, "pricing": { "prompt": "0.016", "completion": "0", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-mini-transcribe@openai", "name": "OpenAI: GPT-4o Mini Transcribe", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000125" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-mini-transcribe", "canonicalSlug": "openai/gpt-4o-mini-transcribe", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o Mini Transcribe is OpenAI's smaller, cost-efficient speech-to-text model built on GPT-4o Mini audio capabilities. It's priced per token (input and output), making it suitable for high-volume transcription workflows that...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4o-mini-transcribe", "model_id": "openai/gpt-4o-mini-transcribe", "model_name": "OpenAI: GPT-4o Mini Transcribe", "context_length": 128000, "pricing": { "prompt": "0.00000125", "completion": "0.000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/whisper-large-v3@deepinfra", "name": "OpenAI: Whisper Large V3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000075" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/whisper-large-v3", "canonicalSlug": "openai/whisper-large-v3", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "GPT", "instruct_type": null }, "description": "Whisper Large V3 is OpenAI's open-source automatic speech recognition model offering both audio transcription and translation. It supports 99+ languages and accepts common audio formats including mp3, mp4, wav, webm,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | openai/whisper-large-v3", "model_id": "openai/whisper-large-v3", "model_name": "OpenAI: Whisper Large V3", "context_length": 0, "pricing": { "prompt": "0.0000075", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9818824168856, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/whisper-large-v3@together", "name": "OpenAI: Whisper Large V3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1500, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/whisper-large-v3", "canonicalSlug": "openai/whisper-large-v3", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "GPT", "instruct_type": null }, "description": "Whisper Large V3 is OpenAI's open-source automatic speech recognition model offering both audio transcription and translation. It supports 99+ languages and accepts common audio formats including mp3, mp4, wav, webm,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | openai/whisper-large-v3", "model_id": "openai/whisper-large-v3", "model_name": "OpenAI: Whisper Large V3", "context_length": 0, "pricing": { "prompt": "0.0015", "completion": "0", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 98.8197424892704, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/whisper-large-v3@groq", "name": "OpenAI: Whisper Large V3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 111000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.111" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/whisper-large-v3", "canonicalSlug": "openai/whisper-large-v3", "servingProvider": "Groq", "servingProviderSlug": "groq", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "seed", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "GPT", "instruct_type": null }, "description": "Whisper Large V3 is OpenAI's open-source automatic speech recognition model offering both audio transcription and translation. It supports 99+ languages and accepts common audio formats including mp3, mp4, wav, webm,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Groq | openai/whisper-large-v3", "model_id": "openai/whisper-large-v3", "model_name": "OpenAI: Whisper Large V3", "context_length": 0, "pricing": { "prompt": "0.111", "completion": "0", "discount": 0 }, "provider_name": "Groq", "tag": "groq", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "seed", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99675327310655, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/whisper-large-v3-turbo@deepinfra", "name": "OpenAI: Whisper Large V3 Turbo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000333" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/whisper-large-v3-turbo", "canonicalSlug": "openai/whisper-large-v3-turbo", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "GPT", "instruct_type": null }, "description": "Whisper Large V3 Turbo is an optimized version of OpenAI's Whisper Large V3 speech recognition model, designed for speed and cost efficiency. It supports transcription across 99+ languages with a...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | openai/whisper-large-v3-turbo", "model_id": "openai/whisper-large-v3-turbo", "model_name": "OpenAI: Whisper Large V3 Turbo", "context_length": 0, "pricing": { "prompt": "0.00000333", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.95951534079408, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/whisper-large-v3-turbo@groq", "name": "OpenAI: Whisper Large V3 Turbo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 40000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.04" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/whisper-large-v3-turbo", "canonicalSlug": "openai/whisper-large-v3-turbo", "servingProvider": "Groq", "servingProviderSlug": "groq", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "seed", "response_format" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "GPT", "instruct_type": null }, "description": "Whisper Large V3 Turbo is an optimized version of OpenAI's Whisper Large V3 speech recognition model, designed for speed and cost efficiency. It supports transcription across 99+ languages with a...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Groq | openai/whisper-large-v3-turbo", "model_id": "openai/whisper-large-v3-turbo", "model_name": "OpenAI: Whisper Large V3 Turbo", "context_length": 0, "pricing": { "prompt": "0.04", "completion": "0", "discount": 0 }, "provider_name": "Groq", "tag": "groq", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "seed", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99910781995807, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.3@xai", "name": "SpaceXAI: Grok 4.3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.3", "canonicalSlug": "x-ai/grok-4.3-20260430", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 1000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...", "uptimeLast30m": 99.95793522023916, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.3-20260430", "model_id": "x-ai/grok-4.3", "model_name": "SpaceXAI: Grok 4.3", "context_length": 1000000, "pricing": { "prompt": "0.00000125", "completion": "0.0000025", "web_search": "0.005", "input_cache_read": "0.0000002", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000005", "input_cache_read": "0.0000004" } ] }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.95793522023916, "uptime_last_5m": 100, "uptime_last_1d": 99.68445654324817, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.3@xai", "name": "SpaceXAI: Grok 4.3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.3", "canonicalSlug": "x-ai/grok-4.3-20260430", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 1000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...", "uptimeLast30m": 99.95793522023916, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.3-20260430", "model_id": "x-ai/grok-4.3", "model_name": "SpaceXAI: Grok 4.3", "context_length": 1000000, "pricing": { "prompt": "0.0000025", "completion": "0.000005", "web_search": "0.005", "input_cache_read": "0.0000004", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000005", "completion": "0.00001", "input_cache_read": "0.0000008" } ] }, "provider_name": "xAI", "tag": "xai/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.95793522023916, "uptime_last_5m": 100, "uptime_last_1d": 99.68445654324817, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.3@xai", "name": "SpaceXAI: Grok 4.3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.3", "canonicalSlug": "x-ai/grok-4.3-20260430", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 1000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...", "uptimeLast30m": 99.98101385988228, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.3-20260430", "model_id": "x-ai/grok-4.3", "model_name": "SpaceXAI: Grok 4.3", "context_length": 1000000, "pricing": { "prompt": "0.00000125", "completion": "0.0000025", "web_search": "0.005", "input_cache_read": "0.0000002", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000005", "input_cache_read": "0.0000004" } ] }, "provider_name": "xAI", "tag": "xai/zdr", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.98101385988228, "uptime_last_5m": 100, "uptime_last_1d": 99.8206589714188, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.3@xai", "name": "SpaceXAI: Grok 4.3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.3", "canonicalSlug": "x-ai/grok-4.3-20260430", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 1000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...", "uptimeLast30m": 99.98101385988228, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.3-20260430", "model_id": "x-ai/grok-4.3", "model_name": "SpaceXAI: Grok 4.3", "context_length": 1000000, "pricing": { "prompt": "0.0000025", "completion": "0.000005", "web_search": "0.005", "input_cache_read": "0.0000004", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000005", "completion": "0.00001", "input_cache_read": "0.0000008" } ] }, "provider_name": "xAI", "tag": "xai/zdr/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.98101385988228, "uptime_last_5m": 100, "uptime_last_1d": 99.8206589714188, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "ibm-granite/granite-4.1-8b@coreweave", "name": "IBM: Granite 4.1 8B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "ibm-granite/granite-4.1-8b", "canonicalSlug": "ibm-granite/granite-4.1-8b-20260429", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | ibm-granite/granite-4.1-8b-20260429", "model_id": "ibm-granite/granite-4.1-8b", "model_name": "IBM: Granite 4.1 8B", "context_length": 131072, "pricing": { "prompt": "0.00000005", "completion": "0.0000001", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-medium-3-5@mistral", "name": "Mistral: Mistral Medium 3.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-medium-3-5", "canonicalSlug": "mistralai/mistral-medium-3.5-20260430", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/mistral-medium-3.5-20260430", "model_id": "mistralai/mistral-medium-3-5", "model_name": "Mistral: Mistral Medium 3.5", "context_length": 262144, "pricing": { "prompt": "0.0000015", "completion": "0.0000075", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96759753741284, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "kwaivgi/kling-v3.0-pro@atlascloud", "name": "Kling: Video v3.0 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "kwaivgi/kling-v3.0-pro", "canonicalSlug": "kwaivgi/kling-v3.0-pro-20260429", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kling v3.0 Pro is Kuaishou's premium video generation model, offering higher visual quality than the Standard tier. It supports text-to-video and image-to-video workflows, with first-frame and last-frame control for precise...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | kwaivgi/kling-v3.0-pro-20260429", "model_id": "kwaivgi/kling-v3.0-pro", "model_name": "Kling: Video v3.0 Pro", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "kwaivgi/kling-v3.0-std@atlascloud", "name": "Kling: Video v3.0 Standard", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "kwaivgi/kling-v3.0-std", "canonicalSlug": "kwaivgi/kling-v3.0-std-20260429", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kling v3.0 Standard is a video generation model from Kuaishou. It supports text-to-video and image-to-video workflows, with first-frame and last-frame control for guided scene composition. Clips range from 3 to...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | kwaivgi/kling-v3.0-std-20260429", "model_id": "kwaivgi/kling-v3.0-std", "model_name": "Kling: Video v3.0 Standard", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free@nvidia", "name": "NVIDIA: Nemotron 3 Nano Omni", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", "canonicalSlug": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-20260428", "servingProvider": "Nvidia", "servingProviderSlug": "nvidia", "contextLength": 256000, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tool_choice", "tools" ], "architecture": { "modality": "text+image+audio+video->text", "input_modalities": [ "text", "audio", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron™ 3 Nano Omni is a 30B-A3B open multimodal model designed to function as a perception and context sub-agent in enterprise agent systems. It accepts text, image, video, and...", "uptimeLast30m": 85.61064087061668, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nvidia | nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-20260428:free", "model_id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", "model_name": "NVIDIA: Nemotron 3 Nano Omni", "context_length": 256000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Nvidia", "tag": "nvidia", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tool_choice", "tools" ], "status": -2, "uptime_last_30m": 85.61064087061668, "uptime_last_5m": 87.5, "uptime_last_1d": 95.91925896532513, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/whisper-1@openai", "name": "OpenAI: Whisper 1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 6000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.006" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/whisper-1", "canonicalSlug": "openai/whisper-1", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "GPT", "instruct_type": null }, "description": "Whisper is OpenAI's open-source automatic speech recognition model, available via API as `whisper-1`. It supports transcription and translation across 50+ languages from audio files up to 25 MB. Accepts formats...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/whisper-1", "model_id": "openai/whisper-1", "model_name": "OpenAI: Whisper 1", "context_length": 0, "pricing": { "prompt": "0.006", "completion": "0", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-transcribe@openai", "name": "OpenAI: GPT-4o Transcribe", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-transcribe", "canonicalSlug": "openai/gpt-4o-transcribe", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "audio->transcription", "input_modalities": [ "audio" ], "output_modalities": [ "transcription" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o Transcribe is OpenAI's high-quality speech-to-text model built on GPT-4o audio capabilities. It's priced per token (input and output), making it suitable for workflows that benefit from token-level billing transparency.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4o-transcribe", "model_id": "openai/gpt-4o-transcribe", "model_name": "OpenAI: GPT-4o Transcribe", "context_length": 128000, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "~anthropic/claude-haiku-latest", "name": "Anthropic Claude Haiku Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "canonicalSlug": "~anthropic/claude-haiku-latest", "contextLength": 200000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_p" ], "description": "This model always redirects to the latest model in the Anthropic Claude Haiku family.", "endpointCount": 0 } }, { "id": "~openai/gpt-mini-latest", "name": "OpenAI GPT Mini Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "canonicalSlug": "~openai/gpt-mini-latest", "contextLength": 400000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "description": "This model always redirects to the latest model in the OpenAI GPT Mini family.", "endpointCount": 0 } }, { "id": "~google/gemini-pro-latest", "name": "Google Gemini Pro Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.000002, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000002" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ] } ], "metadata": { "source": "openrouter", "canonicalSlug": "~google/gemini-pro-latest", "contextLength": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "audio", "file", "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "description": "This model always redirects to the latest model in the Google Gemini Pro family.", "endpointCount": 0 } }, { "id": "~moonshotai/kimi-latest", "name": "MoonshotAI Kimi Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000026" } ], "output": [ { "amount": 13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000013" } ], "cacheRead": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000029" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "canonicalSlug": "~moonshotai/kimi-latest", "contextLength": 1048576, "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "description": "This model always redirects to the latest model in the MoonshotAI Kimi family.", "endpointCount": 0 } }, { "id": "~google/gemini-flash-latest", "name": "Google Gemini Flash Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "output": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "cacheWrite": [ { "amount": 0.0208333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000208333333333333" } ], "other": [ { "amount": 3.75e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000375" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ] } ], "metadata": { "source": "openrouter", "canonicalSlug": "~google/gemini-flash-latest", "contextLength": 1048576, "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "include_reasoning", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_p" ], "description": "This model always redirects to the latest model in the Google Gemini Flash family.", "endpointCount": 0 } }, { "id": "~anthropic/claude-sonnet-latest", "name": "Anthropic Claude Sonnet Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "canonicalSlug": "~anthropic/claude-sonnet-latest", "contextLength": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "tool_choice", "tools", "verbosity" ], "description": "This model always redirects to the latest model in the Anthropic Claude Sonnet family.", "endpointCount": 0 } }, { "id": "~openai/gpt-latest", "name": "OpenAI GPT Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [ { "amount": 3.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "canonicalSlug": "~openai/gpt-latest", "contextLength": 1050000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "seed", "structured_outputs", "tool_choice", "tools" ], "description": "This model always redirects to the latest model in the OpenAI GPT family.", "endpointCount": 0 } }, { "id": "qwen/qwen3.5-plus-20260420@alibaba", "name": "Qwen: Qwen3.5 Plus 2026-04-20", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-plus-20260420", "canonicalSlug": "qwen/qwen3.5-plus-20260420", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.5-plus-20260420", "model_id": "qwen/qwen3.5-plus-20260420", "model_name": "Qwen: Qwen3.5 Plus 2026-04-20", "context_length": 1000000, "pricing": { "prompt": "0.0000003", "completion": "0.0000018", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.000000375", "completion": "0.00000225", "input_cache_write": "0.00000046875" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 983616, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-flash@alibaba", "name": "Qwen: Qwen3.6 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "output": [ { "amount": 1.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001125" } ], "cacheRead": [], "cacheWrite": [ { "amount": 0.234375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000234375" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-flash", "canonicalSlug": "qwen/qwen3.6-flash", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.6-flash", "model_id": "qwen/qwen3.6-flash", "model_name": "Qwen: Qwen3.6 Flash", "context_length": 1000000, "pricing": { "prompt": "0.0000001875", "completion": "0.000001125", "input_cache_write": "0.000000234375", "discount": 0, "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.00000075", "completion": "0.000003", "input_cache_write": "0.0000009375" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 983616, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-35b-a3b@venice", "name": "Qwen: Qwen3.6 35B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.098, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000098" } ], "output": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-35b-a3b", "canonicalSlug": "qwen/qwen3.6-35b-a3b-20260415", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | qwen/qwen3.6-35b-a3b-20260415", "model_id": "qwen/qwen3.6-35b-a3b", "model_name": "Qwen: Qwen3.6 35B A3B", "context_length": 256000, "pricing": { "prompt": "0.000000098", "completion": "0.00000095", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.8637831859899, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-35b-a3b@deepinfra", "name": "Qwen: Qwen3.6 35B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-35b-a3b", "canonicalSlug": "qwen/qwen3.6-35b-a3b-20260415", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...", "uptimeLast30m": 99.11788291900562, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3.6-35b-a3b-20260415", "model_id": "qwen/qwen3.6-35b-a3b", "model_name": "Qwen: Qwen3.6 35B A3B", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.00000095", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.11788291900562, "uptime_last_5m": 100, "uptime_last_1d": 97.53546902778261, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-35b-a3b@akashml", "name": "Qwen: Qwen3.6 35B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-35b-a3b", "canonicalSlug": "qwen/qwen3.6-35b-a3b-20260415", "servingProvider": "AkashML", "servingProviderSlug": "akashml", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...", "uptimeLast30m": 99.31761316248388, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AkashML | qwen/qwen3.6-35b-a3b-20260415", "model_id": "qwen/qwen3.6-35b-a3b", "model_name": "Qwen: Qwen3.6 35B A3B", "context_length": 262144, "pricing": { "prompt": "0.00000014", "completion": "0.000001", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "AkashML", "tag": "akashml/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.31761316248388, "uptime_last_5m": 100, "uptime_last_1d": 98.79969924947436, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-35b-a3b@parasail", "name": "Qwen: Qwen3.6 35B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-35b-a3b", "canonicalSlug": "qwen/qwen3.6-35b-a3b-20260415", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...", "uptimeLast30m": 99.93476842791911, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | qwen/qwen3.6-35b-a3b-20260415", "model_id": "qwen/qwen3.6-35b-a3b", "model_name": "Qwen: Qwen3.6 35B A3B", "context_length": 262144, "pricing": { "prompt": "0.00000015", "completion": "0.000001", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.93476842791911, "uptime_last_5m": 100, "uptime_last_1d": 97.92127404399157, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-35b-a3b@atlascloud", "name": "Qwen: Qwen3.6 35B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.186, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000186" } ], "output": [ { "amount": 1.11375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000111375" } ], "cacheRead": [ { "amount": 0.186, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000186" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-35b-a3b", "canonicalSlug": "qwen/qwen3.6-35b-a3b-20260415", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...", "uptimeLast30m": 99.96749024707412, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | qwen/qwen3.6-35b-a3b-20260415", "model_id": "qwen/qwen3.6-35b-a3b", "model_name": "Qwen: Qwen3.6 35B A3B", "context_length": 262144, "pricing": { "prompt": "0.000000186", "completion": "0.00000111375", "input_cache_read": "0.000000186", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.96749024707412, "uptime_last_5m": 100, "uptime_last_1d": 99.658858923391, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-35b-a3b@io-net", "name": "Qwen: Qwen3.6 35B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000019" } ], "output": [ { "amount": 1.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000119" } ], "cacheRead": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-35b-a3b", "canonicalSlug": "qwen/qwen3.6-35b-a3b-20260415", "servingProvider": "Io Net", "servingProviderSlug": "io-net", "contextLength": 262140, "maxCompletionTokens": 262140, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...", "uptimeLast30m": 98.42233009708737, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Io Net | qwen/qwen3.6-35b-a3b-20260415", "model_id": "qwen/qwen3.6-35b-a3b", "model_name": "Qwen: Qwen3.6 35B A3B", "context_length": 262140, "pricing": { "prompt": "0.00000019", "completion": "0.00000119", "input_cache_read": "0.00000009", "discount": 0 }, "provider_name": "Io Net", "tag": "io-net/fp8", "quantization": "fp8", "max_completion_tokens": 262140, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 98.42233009708737, "uptime_last_5m": 100, "uptime_last_1d": 98.05529637086873, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-35b-a3b@phala", "name": "Qwen: Qwen3.6 35B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000127" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-35b-a3b", "canonicalSlug": "qwen/qwen3.6-35b-a3b-20260415", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "structured_outputs", "tools", "logprobs", "top_logprobs", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | qwen/qwen3.6-35b-a3b-20260415", "model_id": "qwen/qwen3.6-35b-a3b", "model_name": "Qwen: Qwen3.6 35B A3B", "context_length": 262144, "pricing": { "prompt": "0.0000002", "completion": "0.00000127", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "structured_outputs", "tools", "logprobs", "top_logprobs", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.49518794261849, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-35b-a3b@siliconflow", "name": "Qwen: Qwen3.6 35B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-35b-a3b", "canonicalSlug": "qwen/qwen3.6-35b-a3b-20260415", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...", "uptimeLast30m": 98.87766554433222, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3.6-35b-a3b-20260415", "model_id": "qwen/qwen3.6-35b-a3b", "model_name": "Qwen: Qwen3.6 35B A3B", "context_length": 262144, "pricing": { "prompt": "0.0000002", "completion": "0.0000016", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 98.87766554433222, "uptime_last_5m": 100, "uptime_last_1d": 98.74301675977654, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-35b-a3b@coreweave", "name": "Qwen: Qwen3.6 35B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-35b-a3b", "canonicalSlug": "qwen/qwen3.6-35b-a3b-20260415", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...", "uptimeLast30m": 99.90645463049579, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | qwen/qwen3.6-35b-a3b-20260415", "model_id": "qwen/qwen3.6-35b-a3b", "model_name": "Qwen: Qwen3.6 35B A3B", "context_length": 262144, "pricing": { "prompt": "0.00000025", "completion": "0.00000125", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.90645463049579, "uptime_last_5m": 99.32885906040269, "uptime_last_1d": 99.60483668062811, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-max-preview@alibaba", "name": "Qwen: Qwen3.6 Max Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.0270000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001027" } ], "output": [ { "amount": 6.162, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006162" } ], "cacheRead": [], "cacheWrite": [ { "amount": 1.28375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000128375" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-max-preview", "canonicalSlug": "qwen/qwen3.6-max-preview-20260420", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "top_logprobs", "logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.6-max-preview-20260420", "model_id": "qwen/qwen3.6-max-preview", "model_name": "Qwen: Qwen3.6 Max Preview", "context_length": 262144, "pricing": { "prompt": "0.000001027", "completion": "0.000006162", "input_cache_write": "0.00000128375", "discount": 0, "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.00000158", "completion": "0.00000948", "input_cache_write": "0.000001975" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 229376, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "top_logprobs", "logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-27b@chutes", "name": "Qwen: Qwen3.6 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-27b", "canonicalSlug": "qwen/qwen3.6-27b-20260422", "servingProvider": "Chutes", "servingProviderSlug": "chutes", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...", "uptimeLast30m": 99.47089947089947, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Chutes | qwen/qwen3.6-27b-20260422", "model_id": "qwen/qwen3.6-27b", "model_name": "Qwen: Qwen3.6 27B", "context_length": 262144, "pricing": { "prompt": "0.0000003", "completion": "0.000002", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Chutes", "tag": "chutes/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.47089947089947, "uptime_last_5m": 100, "uptime_last_1d": 93.84703124009886, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-27b@siliconflow", "name": "Qwen: Qwen3.6 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-27b", "canonicalSlug": "qwen/qwen3.6-27b-20260422", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "response_format", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...", "uptimeLast30m": 93.56725146198829, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3.6-27b-20260422", "model_id": "qwen/qwen3.6-27b", "model_name": "Qwen: Qwen3.6 27B", "context_length": 262144, "pricing": { "prompt": "0.0000003", "completion": "0.0000032", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "response_format", "tools", "tool_choice", "structured_outputs" ], "status": -2, "uptime_last_30m": 93.56725146198829, "uptime_last_5m": null, "uptime_last_1d": 95.38164737460157, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-27b@phala", "name": "Qwen: Qwen3.6 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000032" } ], "output": [ { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-27b", "canonicalSlug": "qwen/qwen3.6-27b-20260422", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 262144, "maxCompletionTokens": 262140, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | qwen/qwen3.6-27b-20260422", "model_id": "qwen/qwen3.6-27b", "model_name": "Qwen: Qwen3.6 27B", "context_length": 262144, "pricing": { "prompt": "0.00000032", "completion": "0.0000027", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 262140, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.18742801243314, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-27b@deepinfra", "name": "Qwen: Qwen3.6 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000032" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-27b", "canonicalSlug": "qwen/qwen3.6-27b-20260422", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 81920, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3.6-27b-20260422", "model_id": "qwen/qwen3.6-27b", "model_name": "Qwen: Qwen3.6 27B", "context_length": 262144, "pricing": { "prompt": "0.00000032", "completion": "0.0000032", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 81920, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.15220663972744, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-27b@venice", "name": "Qwen: Qwen3.6 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.325, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000325" } ], "output": [ { "amount": 3.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000325" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-27b", "canonicalSlug": "qwen/qwen3.6-27b-20260422", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tool_choice", "tools", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | qwen/qwen3.6-27b-20260422", "model_id": "qwen/qwen3.6-27b", "model_name": "Qwen: Qwen3.6 27B", "context_length": 256000, "pricing": { "prompt": "0.000000325", "completion": "0.00000325", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tool_choice", "tools", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 96.31741346927824, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-27b@alibaba", "name": "Qwen: Qwen3.6 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-27b", "canonicalSlug": "qwen/qwen3.6-27b-20260422", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "top_logprobs", "logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.6-27b-20260422", "model_id": "qwen/qwen3.6-27b", "model_name": "Qwen: Qwen3.6 27B", "context_length": 262144, "pricing": { "prompt": "0.00000045", "completion": "0.0000027", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 258048, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "top_logprobs", "logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.96594060690595, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-27b@coreweave", "name": "Qwen: Qwen3.6 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-27b", "canonicalSlug": "qwen/qwen3.6-27b-20260422", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "tools", "structured_outputs", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | qwen/qwen3.6-27b-20260422", "model_id": "qwen/qwen3.6-27b", "model_name": "Qwen: Qwen3.6 27B", "context_length": 262144, "pricing": { "prompt": "0.0000006", "completion": "0.0000036", "input_cache_read": "0.00000012", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "tools", "structured_outputs", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.84422008966264, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5-pro@openai", "name": "OpenAI: GPT-5.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5-pro", "canonicalSlug": "openai/gpt-5.5-pro-20260423", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.5-pro-20260423", "model_id": "openai/gpt-5.5-pro", "model_name": "OpenAI: GPT-5.5 Pro", "context_length": 1050000, "pricing": { "prompt": "0.00003", "completion": "0.00018", "web_search": "0.01", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00006", "completion": "0.00027" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5-pro@openai", "name": "OpenAI: GPT-5.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5-pro", "canonicalSlug": "openai/gpt-5.5-pro-20260423", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.5-pro-20260423", "model_id": "openai/gpt-5.5-pro", "model_name": "OpenAI: GPT-5.5 Pro", "context_length": 1050000, "pricing": { "prompt": "0.000015", "completion": "0.00009", "web_search": "0.01", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00003", "completion": "0.000135" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5-pro:batch@openai", "name": "OpenAI: GPT-5.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5-pro:batch", "canonicalSlug": "openai/gpt-5.5-pro-20260423", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.5-pro-20260423:batch", "model_id": "openai/gpt-5.5-pro:batch", "model_name": "OpenAI: GPT-5.5 Pro", "context_length": 1050000, "pricing": { "prompt": "0.000015", "completion": "0.00009", "web_search": "0.01", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00003", "completion": "0.000135" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5-pro:batch@openai", "name": "OpenAI: GPT-5.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "output": [ { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5-pro:batch", "canonicalSlug": "openai/gpt-5.5-pro-20260423", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.5-pro-20260423:batch", "model_id": "openai/gpt-5.5-pro:batch", "model_name": "OpenAI: GPT-5.5 Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000075", "completion": "0.000045", "web_search": "0.01", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000015", "completion": "0.0000675" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5@openai", "name": "OpenAI: GPT-5.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5", "canonicalSlug": "openai/gpt-5.5-20260423", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "uptimeLast30m": 99.82558736021339, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.5-20260423", "model_id": "openai/gpt-5.5", "model_name": "OpenAI: GPT-5.5", "context_length": 1050000, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00001", "completion": "0.000045", "input_cache_read": "0.000001" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.82558736021339, "uptime_last_5m": 99.63547995139733, "uptime_last_1d": 99.1259200363707, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5@openai", "name": "OpenAI: GPT-5.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5", "canonicalSlug": "openai/gpt-5.5-20260423", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "uptimeLast30m": 99.82558736021339, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.5-20260423", "model_id": "openai/gpt-5.5", "model_name": "OpenAI: GPT-5.5", "context_length": 1050000, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000005", "completion": "0.0000225", "input_cache_read": "0.0000005" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.82558736021339, "uptime_last_5m": 99.63547995139733, "uptime_last_1d": 99.1259200363707, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5@openai", "name": "OpenAI: GPT-5.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5", "canonicalSlug": "openai/gpt-5.5-20260423", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "uptimeLast30m": 99.82558736021339, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.5-20260423", "model_id": "openai/gpt-5.5", "model_name": "OpenAI: GPT-5.5", "context_length": 1050000, "pricing": { "prompt": "0.0000125", "completion": "0.000075", "web_search": "0.01", "input_cache_read": "0.00000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.82558736021339, "uptime_last_5m": 99.63547995139733, "uptime_last_1d": 99.1259200363707, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5@azure", "name": "OpenAI: GPT-5.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5", "canonicalSlug": "openai/gpt-5.5-20260423", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.5-20260423", "model_id": "openai/gpt-5.5", "model_name": "OpenAI: GPT-5.5", "context_length": 1050000, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00001", "completion": "0.000045", "input_cache_read": "0.000001" } ] }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.93642545737258, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5@azure", "name": "OpenAI: GPT-5.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5", "canonicalSlug": "openai/gpt-5.5-20260423", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.5-20260423", "model_id": "openai/gpt-5.5", "model_name": "OpenAI: GPT-5.5", "context_length": 1050000, "pricing": { "prompt": "0.0000055", "completion": "0.000033", "web_search": "0.01", "input_cache_read": "0.00000055", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000011", "completion": "0.0000495", "input_cache_read": "0.0000011" } ] }, "provider_name": "Azure", "tag": "azure/eu", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5@azure", "name": "OpenAI: GPT-5.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5", "canonicalSlug": "openai/gpt-5.5-20260423", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.5-20260423", "model_id": "openai/gpt-5.5", "model_name": "OpenAI: GPT-5.5", "context_length": 1050000, "pricing": { "prompt": "0.0000055", "completion": "0.000033", "web_search": "0.01", "input_cache_read": "0.00000055", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000011", "completion": "0.0000495", "input_cache_read": "0.0000011" } ] }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 98.3480825958702, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5@amazon-bedrock", "name": "OpenAI: GPT-5.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5", "canonicalSlug": "openai/gpt-5.5-20260423", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | openai/gpt-5.5-20260423", "model_id": "openai/gpt-5.5", "model_name": "OpenAI: GPT-5.5", "context_length": 1050000, "pricing": { "prompt": "0.0000055", "completion": "0.000033", "web_search": "0.01", "input_cache_read": "0.00000055", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000011", "completion": "0.0000495", "input_cache_read": "0.0000011" } ] }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/us-east-1", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5:batch@openai", "name": "OpenAI: GPT-5.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5:batch", "canonicalSlug": "openai/gpt-5.5-20260423", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.5-20260423:batch", "model_id": "openai/gpt-5.5:batch", "model_name": "OpenAI: GPT-5.5", "context_length": 1050000, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000005", "completion": "0.0000225", "input_cache_read": "0.0000005" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5:batch@openai", "name": "OpenAI: GPT-5.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5:batch", "canonicalSlug": "openai/gpt-5.5-20260423", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.5-20260423:batch", "model_id": "openai/gpt-5.5:batch", "model_name": "OpenAI: GPT-5.5", "context_length": 1050000, "pricing": { "prompt": "0.00000125", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.000000125", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000025", "completion": "0.00001125", "input_cache_read": "0.00000025" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.5:batch@openai", "name": "OpenAI: GPT-5.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "output": [ { "amount": 37.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000375" } ], "cacheRead": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.5:batch", "canonicalSlug": "openai/gpt-5.5-20260423", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.5-20260423:batch", "model_id": "openai/gpt-5.5:batch", "model_name": "OpenAI: GPT-5.5", "context_length": 1050000, "pricing": { "prompt": "0.00000625", "completion": "0.0000375", "web_search": "0.01", "input_cache_read": "0.000000625", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@deepseek", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000066" } ], "output": [ { "amount": 1.9800000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000198" } ], "cacheRead": [ { "amount": 0.022, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000022" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "DeepSeek", "servingProviderSlug": "deepseek", "contextLength": 1048576, "maxCompletionTokens": 384000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 99.99049248906637, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepSeek | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000066", "completion": "0.00000198", "input_cache_read": "0.000000022", "discount": 0, "overrides": [ { "utc_start": 1000, "utc_end": 100, "prompt": "0.00000066", "completion": "0.00000198", "input_cache_read": "0.000000022" }, { "utc_start": 100, "utc_end": 400, "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044" }, { "utc_start": 400, "utc_end": 600, "prompt": "0.00000066", "completion": "0.00000198", "input_cache_read": "0.000000022" }, { "utc_start": 600, "utc_end": 1000, "prompt": "0.00000132", "completion": "0.00000396", "input_cache_read": "0.000000044" } ] }, "provider_name": "DeepSeek", "tag": "deepseek", "quantization": "unknown", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.99049248906637, "uptime_last_5m": 100, "uptime_last_1d": 98.01867790295795, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@streamlake", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.69426, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000069426" } ], "output": [ { "amount": 1.38852, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138852" } ], "cacheRead": [ { "amount": 0.057855, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000057855" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 1024000, "maxCompletionTokens": 384000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs", "structured_outputs", "max_tokens", "temperature", "top_p", "stop", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 98.52899708619232, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1024000, "pricing": { "prompt": "0.00000069426", "completion": "0.00000138852", "input_cache_read": "0.000000057855", "discount": 0.601 }, "provider_name": "StreamLake", "tag": "streamlake/fp8", "quantization": "fp8", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs", "structured_outputs", "max_tokens", "temperature", "top_p", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.52899708619232, "uptime_last_5m": 98.50993377483444, "uptime_last_1d": 96.4633135160569, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@gmicloud", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.696, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000696" } ], "output": [ { "amount": 1.392, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001392" } ], "cacheRead": [ { "amount": 0.058, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000058" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 96.63355408388522, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.000000696", "completion": "0.000001392", "input_cache_read": "0.000000058", "discount": 0.6 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 96.63355408388522, "uptime_last_5m": 98.72068230277186, "uptime_last_1d": 94.4663973901793, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@digitalocean", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.87, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000087" } ], "output": [ { "amount": 1.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000174" } ], "cacheRead": [ { "amount": 0.174, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000174" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 98.83268482490273, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000087", "completion": "0.00000174", "input_cache_read": "0.000000174", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.83268482490273, "uptime_last_5m": 98.0349344978166, "uptime_last_1d": 92.96453175481204, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@ionstream", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.131, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001131" } ], "output": [ { "amount": 2.262, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002262" } ], "cacheRead": [ { "amount": 0.094, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000094" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "Ionstream", "servingProviderSlug": "ionstream", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 99.56627342123525, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Ionstream | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.000001131", "completion": "0.000002262", "input_cache_read": "0.000000094", "discount": 0 }, "provider_name": "Ionstream", "tag": "ionstream/fp4", "quantization": "fp4", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.56627342123525, "uptime_last_5m": 98.5239852398524, "uptime_last_1d": 96.90960334114624, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@coreweave", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000115" } ], "output": [ { "amount": 2.5500000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000255" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000115", "completion": "0.00000255", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp8", "quantization": "fp8", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 93.9079333986288, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@deepinfra", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000013" } ], "output": [ { "amount": 2.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000026" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 1048576, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 99.75247524752476, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.0000013", "completion": "0.0000026", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.75247524752476, "uptime_last_5m": 98.91304347826086, "uptime_last_1d": 98.42329808940828, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@alibaba", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4160000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001416" } ], "output": [ { "amount": 2.8320000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002832" } ], "cacheRead": [ { "amount": 0.118, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000118" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 393216, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "structured_outputs", "top_k", "frequency_penalty", "stop", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 99.76672805402087, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1000000, "pricing": { "prompt": "0.000001416", "completion": "0.000002832", "input_cache_read": "0.000000118", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba/fp8", "quantization": "fp8", "max_completion_tokens": 393216, "max_prompt_tokens": 1000000, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "structured_outputs", "top_k", "frequency_penalty", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.76672805402087, "uptime_last_5m": 99.79654120040692, "uptime_last_1d": 97.87095088819227, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@novita", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "output": [ { "amount": 2.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000288" } ], "cacheRead": [ { "amount": 0.1215, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001215" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 99.84004265529192, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000144", "completion": "0.00000288", "input_cache_read": "0.0000001215", "discount": 0.1 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.84004265529192, "uptime_last_5m": 99.8003992015968, "uptime_last_1d": 99.51310568969726, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@siliconflow", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.5016200000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000150162" } ], "output": [ { "amount": 3.1350000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003135" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000135" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 99.05897114178168, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000150162", "completion": "0.000003135", "input_cache_read": "0.000000135", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.05897114178168, "uptime_last_5m": 98.4375, "uptime_last_1d": 96.70863279545242, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@venice", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.6500000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000165" } ], "output": [ { "amount": 3.3009999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003301" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000033" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 1000000, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 98.29977628635346, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1000000, "pricing": { "prompt": "0.00000165", "completion": "0.000003301", "input_cache_read": "0.00000033", "discount": 0 }, "provider_name": "Venice", "tag": "venice", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.29977628635346, "uptime_last_5m": 95, "uptime_last_1d": 61.038055389519826, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@atlascloud", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000168" } ], "output": [ { "amount": 3.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000338" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "tools", "structured_outputs", "response_format", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 98.37925445705025, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000168", "completion": "0.00000338", "input_cache_read": "0.00000013", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp4", "quantization": "fp4", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "tools", "structured_outputs", "response_format", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.37925445705025, "uptime_last_5m": 98.4126984126984, "uptime_last_1d": 91.14414320108867, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@baidu", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.69, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000169" } ], "output": [ { "amount": 3.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000338" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "Baidu", "servingProviderSlug": "baidu", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 98.8880063542494, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Baidu | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000169", "completion": "0.00000338", "input_cache_read": "0.00000014", "discount": 0 }, "provider_name": "Baidu", "tag": "baidu/fp8", "quantization": "fp8", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.8880063542494, "uptime_last_5m": 98.61111111111111, "uptime_last_1d": 97.25308395278822, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@parasail", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000174" } ], "output": [ { "amount": 3.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000348" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000174", "completion": "0.00000348", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 93.2033008252063, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@baseten", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000174" } ], "output": [ { "amount": 3.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000348" } ], "cacheRead": [ { "amount": 0.145, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000145" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "BaseTen", "servingProviderSlug": "baseten", "contextLength": 1048576, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 99.80670103092784, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "BaseTen | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000174", "completion": "0.00000348", "input_cache_read": "0.000000145", "discount": 0 }, "provider_name": "BaseTen", "tag": "baseten/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.80670103092784, "uptime_last_5m": 100, "uptime_last_1d": 99.22852330977473, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@together", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000174" } ], "output": [ { "amount": 3.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000348" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 512000, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "structured_outputs", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 87.8062678062678, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 512000, "pricing": { "prompt": "0.00000174", "completion": "0.00000348", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "structured_outputs", "response_format", "reasoning_effort" ], "status": -2, "uptime_last_30m": 87.8062678062678, "uptime_last_5m": 98.85057471264368, "uptime_last_1d": 82.68927369841663, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@fireworks", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000174" } ], "output": [ { "amount": 3.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000348" } ], "cacheRead": [ { "amount": 0.145, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000145" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "Fireworks", "servingProviderSlug": "fireworks", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fireworks | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000174", "completion": "0.00000348", "input_cache_read": "0.000000145", "discount": 0 }, "provider_name": "Fireworks", "tag": "fireworks", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-pro@azure", "name": "DeepSeek: DeepSeek V4 Pro 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.91, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000191" } ], "output": [ { "amount": 3.8299999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000383" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-pro", "canonicalSlug": "deepseek/deepseek-v4-pro-20260423", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1048576, "maxCompletionTokens": 384000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...", "uptimeLast30m": 99.01960784313727, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | deepseek/deepseek-v4-pro-20260423", "model_id": "deepseek/deepseek-v4-pro", "model_name": "DeepSeek: DeepSeek V4 Pro 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000191", "completion": "0.00000383", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.01960784313727, "uptime_last_5m": 97.12230215827337, "uptime_last_1d": 95.2928198465423, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@digitalocean", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.0679, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000679" } ], "output": [ { "amount": 0.16799999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000168" } ], "cacheRead": [ { "amount": 0.016800000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000168" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "tool_choice", "tools", "structured_outputs", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.08213022277465, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.0000000679", "completion": "0.000000168", "input_cache_read": "0.0000000168", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "tool_choice", "tools", "structured_outputs", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.08213022277465, "uptime_last_5m": 99.18200408997954, "uptime_last_1d": 98.29845994222241, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@streamlake", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08259999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000826" } ], "output": [ { "amount": 0.16519999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001652" } ], "cacheRead": [ { "amount": 0.01652, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001652" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 1024000, "maxCompletionTokens": 384000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "structured_outputs", "max_tokens", "temperature", "top_p", "stop", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.71355824315722, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1024000, "pricing": { "prompt": "0.0000000826", "completion": "0.0000001652", "input_cache_read": "0.00000001652", "discount": 0.41 }, "provider_name": "StreamLake", "tag": "streamlake/fp8", "quantization": "fp8", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "structured_outputs", "max_tokens", "temperature", "top_p", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.71355824315722, "uptime_last_5m": 99.90125177892017, "uptime_last_1d": 98.44597119858845, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@gmicloud", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08399999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000084" } ], "output": [ { "amount": 0.16799999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000168" } ], "cacheRead": [ { "amount": 0.016800000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000168" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 1048575, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 97.96345122283844, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048575, "pricing": { "prompt": "0.000000084", "completion": "0.000000168", "input_cache_read": "0.0000000168", "discount": 0.4 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.96345122283844, "uptime_last_5m": 99.94974242995352, "uptime_last_1d": 99.17543113761712, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@deepinfra", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "cacheRead": [ { "amount": 0.018, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000018" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "structured_outputs", "logit_bias", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.9200541854965, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000009", "completion": "0.00000018", "input_cache_read": "0.000000018", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "structured_outputs", "logit_bias", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.9200541854965, "uptime_last_5m": 99.99253842710043, "uptime_last_1d": 99.45501169200031, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@sail-research", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "Sail Research", "servingProviderSlug": "sail-research", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.73508985465742, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Sail Research | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000009", "completion": "0.00000018", "input_cache_read": "0.00000002", "discount": 0 }, "provider_name": "Sail Research", "tag": "sail-research/fp4", "quantization": "fp4", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.73508985465742, "uptime_last_5m": 99.8689384010485, "uptime_last_1d": 99.57876804160158, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@siliconflow", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.56996253461476, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000013", "completion": "0.00000028", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.56996253461476, "uptime_last_5m": 99.87772071411104, "uptime_last_1d": 97.71759804967687, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@alibaba", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.134, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000134" } ], "output": [ { "amount": 0.268, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000268" } ], "cacheRead": [ { "amount": 0.026799999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000268" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 393216, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "structured_outputs", "top_k", "frequency_penalty", "stop", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.74069617889889, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1000000, "pricing": { "prompt": "0.000000134", "completion": "0.000000268", "input_cache_read": "0.0000000268", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba/fp8", "quantization": "fp8", "max_completion_tokens": 393216, "max_prompt_tokens": 1000000, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "structured_outputs", "top_k", "frequency_penalty", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.74069617889889, "uptime_last_5m": 99.73154362416108, "uptime_last_1d": 99.68965439965886, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@venice", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13799999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000138" } ], "output": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 1000000, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.89011378540286, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1000000, "pricing": { "prompt": "0.000000138", "completion": "0.000000275", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "Venice", "tag": "venice", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.89011378540286, "uptime_last_5m": 99.95291902071564, "uptime_last_1d": 99.76686533422391, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@parasail", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "structured_outputs", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.8177889996616, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "structured_outputs", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.8177889996616, "uptime_last_5m": 99.80799999999999, "uptime_last_1d": 99.01739654303387, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@novita", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tool_choice", "tools", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.37996652375311, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tool_choice", "tools", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.37996652375311, "uptime_last_5m": 99.97839146030512, "uptime_last_1d": 99.65152266210497, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@atlascloud", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.93162727095137, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp4", "quantization": "fp4", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.93162727095137, "uptime_last_5m": 99.90182328190743, "uptime_last_1d": 99.8971313312416, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@coreweave", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.48180766835691, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp8", "quantization": "fp8", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.48180766835691, "uptime_last_5m": 99.60600837232208, "uptime_last_1d": 99.17273907871119, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@baidu", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "Baidu", "servingProviderSlug": "baidu", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.95814148179154, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Baidu | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "Baidu", "tag": "baidu/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "stop", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.95814148179154, "uptime_last_5m": 99.84276729559748, "uptime_last_1d": 95.86185443649981, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@mancer-2", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.145, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000145" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "Mancer 2", "servingProviderSlug": "mancer-2", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "response_format", "logprobs", "top_logprobs", "tools", "structured_outputs", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.22892392049349, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mancer 2 | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.000000145", "completion": "0.00000045", "discount": 0 }, "provider_name": "Mancer 2", "tag": "mancer/fp8", "quantization": "fp8", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "response_format", "logprobs", "top_logprobs", "tools", "structured_outputs", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.22892392049349, "uptime_last_5m": 100, "uptime_last_1d": 98.34034312687223, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@phala", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 1048576, "maxCompletionTokens": 393216, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 98.87392900856793, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.0000002", "completion": "0.0000004", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 393216, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.87392900856793, "uptime_last_5m": 99.81981981981983, "uptime_last_1d": 96.77052065003623, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@azure", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000021" } ], "output": [ { "amount": 0.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000056" } ], "cacheRead": [ { "amount": 0.031, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000031" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1048576, "maxCompletionTokens": 384000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 95.48677654273668, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000021", "completion": "0.00000056", "input_cache_read": "0.000000031", "discount": 0 }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 95.48677654273668, "uptime_last_5m": 95.8688524590164, "uptime_last_1d": 86.29103509999885, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@deepseek", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000066" } ], "cacheRead": [ { "amount": 0.007, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "DeepSeek", "servingProviderSlug": "deepseek", "contextLength": 1048576, "maxCompletionTokens": 384000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.9291824825789, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepSeek | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 1048576, "pricing": { "prompt": "0.00000022", "completion": "0.00000066", "input_cache_read": "0.000000007", "discount": 0, "overrides": [ { "utc_start": 1000, "utc_end": 100, "prompt": "0.00000022", "completion": "0.00000066", "input_cache_read": "0.000000007" }, { "utc_start": 100, "utc_end": 400, "prompt": "0.00000044", "completion": "0.00000132", "input_cache_read": "0.000000014" }, { "utc_start": 400, "utc_end": 600, "prompt": "0.00000022", "completion": "0.00000066", "input_cache_read": "0.000000007" }, { "utc_start": 600, "utc_end": 1000, "prompt": "0.00000044", "completion": "0.00000132", "input_cache_read": "0.000000014" } ] }, "provider_name": "DeepSeek", "tag": "deepseek", "quantization": "unknown", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.9291824825789, "uptime_last_5m": 100, "uptime_last_1d": 97.8845362440965, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v4-flash@cloudflare", "name": "DeepSeek: DeepSeek V4 Flash 0423", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000044" } ], "output": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "cacheRead": [ { "amount": 0.014, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000014" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v4-flash", "canonicalSlug": "deepseek/deepseek-v4-flash-20260423", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 384000, "maxCompletionTokens": 384000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "structured_outputs", "tool_choice", "tools", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...", "uptimeLast30m": 99.9973150758491, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | deepseek/deepseek-v4-flash-20260423", "model_id": "deepseek/deepseek-v4-flash", "model_name": "DeepSeek: DeepSeek V4 Flash 0423", "context_length": 384000, "pricing": { "prompt": "0.00000044", "completion": "0.00000132", "input_cache_read": "0.000000014", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 384000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "structured_outputs", "tool_choice", "tools", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.9973150758491, "uptime_last_5m": 100, "uptime_last_1d": 99.99661013057012, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-tts-preview@google", "name": "Google: Gemini 3.1 Flash TTS Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-tts-preview", "canonicalSlug": "google/gemini-3.1-flash-tts-preview", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 32768, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash TTS Preview is a text-to-speech model from Google, and a substantial generational step up from Gemini 2.5 Flash TTS. It takes text input and produces audio output...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-flash-tts-preview", "model_id": "google/gemini-3.1-flash-tts-preview", "model_name": "Google: Gemini 3.1 Flash TTS Preview", "context_length": 32768, "pricing": { "prompt": "0.000001", "completion": "0.00002", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": 8192, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/veo-3.1-fast@google", "name": "Google: Veo 3.1 Fast", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/veo-3.1-fast", "canonicalSlug": "google/veo-3.1-fast-20260320", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Google's mid-tier video generation model balancing speed and quality. Veo 3.1 Fast generates high-quality video from text or image prompts with native synchronized audio, offering faster turnaround than Veo 3.1...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/veo-3.1-fast-20260320", "model_id": "google/veo-3.1-fast", "model_name": "Google: Veo 3.1 Fast", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "canopylabs/orpheus-3b-0.1-ft@deepinfra", "name": "Canopy Labs: Orpheus 3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000007" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "canopylabs/orpheus-3b-0.1-ft", "canonicalSlug": "canopylabs/orpheus-3b-0.1-ft", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "Orpheus 3B is an English text-to-speech model from Canopy Labs, fine-tuned for natural prosody and expressive delivery. It offers 7 preset voices and is suited for narration, voice assistants, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | canopylabs/orpheus-3b-0.1-ft", "model_id": "canopylabs/orpheus-3b-0.1-ft", "model_name": "Canopy Labs: Orpheus 3B", "context_length": 4096, "pricing": { "prompt": "0.000007", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "canopylabs/orpheus-3b-0.1-ft@together", "name": "Canopy Labs: Orpheus 3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "canopylabs/orpheus-3b-0.1-ft", "canonicalSlug": "canopylabs/orpheus-3b-0.1-ft", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "Orpheus 3B is an English text-to-speech model from Canopy Labs, fine-tuned for natural prosody and expressive delivery. It offers 7 preset voices and is suited for narration, voice assistants, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | canopylabs/orpheus-3b-0.1-ft", "model_id": "canopylabs/orpheus-3b-0.1-ft", "model_name": "Canopy Labs: Orpheus 3B", "context_length": 4096, "pricing": { "prompt": "0.000015", "completion": "0", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sesame/csm-1b@deepinfra", "name": "Sesame: CSM 1B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000007" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sesame/csm-1b", "canonicalSlug": "sesame/csm-1b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "CSM 1B is a conversational speech model from Sesame. It accepts text input and produces English speech output, with voice options spanning conversational and read-speech styles. At 1B parameters, it...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | sesame/csm-1b", "model_id": "sesame/csm-1b", "model_name": "Sesame: CSM 1B", "context_length": 4096, "pricing": { "prompt": "0.000007", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "hexgrad/kokoro-82m@deepinfra", "name": "hexgrad: Kokoro 82M", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.62, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000062" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "hexgrad/kokoro-82m", "canonicalSlug": "hexgrad/kokoro-82m", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kokoro 82M is a lightweight, open-weight text-to-speech model from hexgrad. It converts text to speech across 8 languages (American and British English, Spanish, French, Hindi, Italian, Japanese, Portuguese, and Chinese)...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | hexgrad/kokoro-82m", "model_id": "hexgrad/kokoro-82m", "model_name": "hexgrad: Kokoro 82M", "context_length": 4096, "pricing": { "prompt": "0.00000062", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99779580320931, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "hexgrad/kokoro-82m@together", "name": "hexgrad: Kokoro 82M", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "hexgrad/kokoro-82m", "canonicalSlug": "hexgrad/kokoro-82m", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kokoro 82M is a lightweight, open-weight text-to-speech model from hexgrad. It converts text to speech across 8 languages (American and British English, Spanish, French, Hindi, Italian, Japanese, Portuguese, and Chinese)...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | hexgrad/kokoro-82m", "model_id": "hexgrad/kokoro-82m", "model_name": "hexgrad: Kokoro 82M", "context_length": 4096, "pricing": { "prompt": "0.000004", "completion": "0", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/veo-3.1-lite@google", "name": "Google: Veo 3.1 Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/veo-3.1-lite", "canonicalSlug": "google/veo-3.1-lite-20260331", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Google's most cost-effective video generation model, designed for high-volume applications and rapid iteration. Veo 3.1 Lite generates 720p and 1080p video from text or image prompts with native synchronized audio...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/veo-3.1-lite-20260331", "model_id": "google/veo-3.1-lite", "model_name": "Google: Veo 3.1 Lite", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "inclusionai/ling-2.6-1t@novita", "name": "inclusionAI: Ling-2.6-1T", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "output": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "inclusionai/ling-2.6-1t", "canonicalSlug": "inclusionai/ling-2.6-1t-20260423", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | inclusionai/ling-2.6-1t-20260423", "model_id": "inclusionai/ling-2.6-1t", "model_name": "inclusionAI: Ling-2.6-1T", "context_length": 262144, "pricing": { "prompt": "0.000000075", "completion": "0.000000625", "input_cache_read": "0.000000015", "discount": 0.75 }, "provider_name": "Novita", "tag": "novita", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "tencent/hy3-preview@gmicloud", "name": "Tencent: Hy3 preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "tencent/hy3-preview", "canonicalSlug": "tencent/hy3-preview-20260421", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 262144, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...", "uptimeLast30m": 99.97669540899558, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | tencent/hy3-preview-20260421", "model_id": "tencent/hy3-preview", "model_name": "Tencent: Hy3 preview", "context_length": 262144, "pricing": { "prompt": "0.00000018", "completion": "0.0000006", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "GMICloud", "tag": "gmicloud/bf16", "quantization": "bf16", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.97669540899558, "uptime_last_5m": 100, "uptime_last_1d": 99.89471755318263, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5-pro@gmicloud", "name": "Xiaomi: MiMo-V2.5-Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003045" } ], "output": [ { "amount": 0.609, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000609" } ], "cacheRead": [ { "amount": 0.0028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5-pro", "canonicalSlug": "xiaomi/mimo-v2.5-pro-20260422", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 1050000, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....", "uptimeLast30m": 98.4313725490196, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | xiaomi/mimo-v2.5-pro-20260422", "model_id": "xiaomi/mimo-v2.5-pro", "model_name": "Xiaomi: MiMo-V2.5-Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000003045", "completion": "0.000000609", "input_cache_read": "0.0000000028", "discount": 0.3 }, "provider_name": "GMICloud", "tag": "gmicloud/bf16", "quantization": "bf16", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": 98.4313725490196, "uptime_last_5m": 100, "uptime_last_1d": 88.85553649704593, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5-pro@digitalocean", "name": "Xiaomi: MiMo-V2.5-Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5-pro", "canonicalSlug": "xiaomi/mimo-v2.5-pro-20260422", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....", "uptimeLast30m": 99.65753424657534, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | xiaomi/mimo-v2.5-pro-20260422", "model_id": "xiaomi/mimo-v2.5-pro", "model_name": "Xiaomi: MiMo-V2.5-Pro", "context_length": 262144, "pricing": { "prompt": "0.0000004", "completion": "0.0000015", "input_cache_read": "0.00000008", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.65753424657534, "uptime_last_5m": 100, "uptime_last_1d": 98.59785202863962, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5-pro@xiaomi", "name": "Xiaomi: MiMo-V2.5-Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.435, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000435" } ], "output": [ { "amount": 0.87, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000087" } ], "cacheRead": [ { "amount": 0.0036, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000036" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5-pro", "canonicalSlug": "xiaomi/mimo-v2.5-pro-20260422", "servingProvider": "Xiaomi", "servingProviderSlug": "xiaomi", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....", "uptimeLast30m": 99.911838790932, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Xiaomi | xiaomi/mimo-v2.5-pro-20260422", "model_id": "xiaomi/mimo-v2.5-pro", "model_name": "Xiaomi: MiMo-V2.5-Pro", "context_length": 1048576, "pricing": { "prompt": "0.000000435", "completion": "0.00000087", "input_cache_read": "0.0000000036", "discount": 0 }, "provider_name": "Xiaomi", "tag": "xiaomi/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.911838790932, "uptime_last_5m": 99.89356040447046, "uptime_last_1d": 98.98972105228239, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5-pro@atlascloud", "name": "Xiaomi: MiMo-V2.5-Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.435, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000435" } ], "output": [ { "amount": 0.87, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000087" } ], "cacheRead": [ { "amount": 0.0036, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000036" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5-pro", "canonicalSlug": "xiaomi/mimo-v2.5-pro-20260422", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 1024000, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....", "uptimeLast30m": 99.14893617021276, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | xiaomi/mimo-v2.5-pro-20260422", "model_id": "xiaomi/mimo-v2.5-pro", "model_name": "Xiaomi: MiMo-V2.5-Pro", "context_length": 1024000, "pricing": { "prompt": "0.000000435", "completion": "0.00000087", "input_cache_read": "0.0000000036", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.14893617021276, "uptime_last_5m": null, "uptime_last_1d": 95.21519145480667, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5-pro@novita", "name": "Xiaomi: MiMo-V2.5-Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.48024, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000048024" } ], "output": [ { "amount": 0.96048, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000096048" } ], "cacheRead": [ { "amount": 0.003956, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000003956" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5-pro", "canonicalSlug": "xiaomi/mimo-v2.5-pro-20260422", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....", "uptimeLast30m": 99.93183367416496, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | xiaomi/mimo-v2.5-pro-20260422", "model_id": "xiaomi/mimo-v2.5-pro", "model_name": "Xiaomi: MiMo-V2.5-Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000048024", "completion": "0.00000096048", "input_cache_read": "0.000000003956", "discount": 0.08 }, "provider_name": "Novita", "tag": "novita", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.93183367416496, "uptime_last_5m": 100, "uptime_last_1d": 93.10167486671352, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5-pro@streamlake", "name": "Xiaomi: MiMo-V2.5-Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.522, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000522" } ], "output": [ { "amount": 1.044, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001044" } ], "cacheRead": [ { "amount": 0.00432, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000000432" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5-pro", "canonicalSlug": "xiaomi/mimo-v2.5-pro-20260422", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "tools", "tool_choice", "max_tokens", "temperature", "top_p", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....", "uptimeLast30m": 98.34983498349835, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | xiaomi/mimo-v2.5-pro-20260422", "model_id": "xiaomi/mimo-v2.5-pro", "model_name": "Xiaomi: MiMo-V2.5-Pro", "context_length": 1000000, "pricing": { "prompt": "0.000000522", "completion": "0.000001044", "input_cache_read": "0.00000000432", "discount": -0.2 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "tools", "tool_choice", "max_tokens", "temperature", "top_p", "stop" ], "status": 0, "uptime_last_30m": 98.34983498349835, "uptime_last_5m": 98.59154929577466, "uptime_last_1d": 93.67970005356186, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5-pro@deepinfra", "name": "Xiaomi: MiMo-V2.5-Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5-pro", "canonicalSlug": "xiaomi/mimo-v2.5-pro-20260422", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 1048576, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | xiaomi/mimo-v2.5-pro-20260422", "model_id": "xiaomi/mimo-v2.5-pro", "model_name": "Xiaomi: MiMo-V2.5-Pro", "context_length": 1048576, "pricing": { "prompt": "0.000001", "completion": "0.000003", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 96.89348262991673, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5@gmicloud", "name": "Xiaomi: MiMo-V2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.119, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000119" } ], "output": [ { "amount": 0.238, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000238" } ], "cacheRead": [ { "amount": 0.00255, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000000255" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5", "canonicalSlug": "xiaomi/mimo-v2.5-20260422", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 1050000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image+audio+video->text", "input_modalities": [ "text", "audio", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...", "uptimeLast30m": 98.5259227567967, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | xiaomi/mimo-v2.5-20260422", "model_id": "xiaomi/mimo-v2.5", "model_name": "Xiaomi: MiMo-V2.5", "context_length": 1050000, "pricing": { "prompt": "0.000000119", "completion": "0.000000238", "input_cache_read": "0.00000000255", "discount": 0.15 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 98.5259227567967, "uptime_last_5m": 95.44452608376194, "uptime_last_1d": 94.02103604145917, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5@xiaomi", "name": "Xiaomi: MiMo-V2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.0028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5", "canonicalSlug": "xiaomi/mimo-v2.5-20260422", "servingProvider": "Xiaomi", "servingProviderSlug": "xiaomi", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "response_format", "tools", "tool_choice", "frequency_penalty", "presence_penalty" ], "architecture": { "modality": "text+image+audio+video->text", "input_modalities": [ "text", "audio", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...", "uptimeLast30m": 99.58849814445482, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Xiaomi | xiaomi/mimo-v2.5-20260422", "model_id": "xiaomi/mimo-v2.5", "model_name": "Xiaomi: MiMo-V2.5", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.0000000028", "discount": 0 }, "provider_name": "Xiaomi", "tag": "xiaomi/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "response_format", "tools", "tool_choice", "frequency_penalty", "presence_penalty" ], "status": 0, "uptime_last_30m": 99.58849814445482, "uptime_last_5m": 99.42363112391931, "uptime_last_1d": 99.17620494563792, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5@parasail", "name": "Xiaomi: MiMo-V2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5", "canonicalSlug": "xiaomi/mimo-v2.5-20260422", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 1048576, "maxCompletionTokens": 1048576, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+audio+video->text", "input_modalities": [ "text", "audio", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...", "uptimeLast30m": 99.59847698165454, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | xiaomi/mimo-v2.5-20260422", "model_id": "xiaomi/mimo-v2.5", "model_name": "Xiaomi: MiMo-V2.5", "context_length": 1048576, "pricing": { "prompt": "0.00000014", "completion": "0.00000028", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 1048576, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.59847698165454, "uptime_last_5m": 99.8859749144812, "uptime_last_1d": 98.34236130574797, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5@novita", "name": "Xiaomi: MiMo-V2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.16799999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000168" } ], "output": [ { "amount": 0.33599999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000336" } ], "cacheRead": [ { "amount": 0.0034, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000034" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5", "canonicalSlug": "xiaomi/mimo-v2.5-20260422", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 1048576, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text+image+audio+video->text", "input_modalities": [ "text", "audio", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...", "uptimeLast30m": 99.93556701030928, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | xiaomi/mimo-v2.5-20260422", "model_id": "xiaomi/mimo-v2.5", "model_name": "Xiaomi: MiMo-V2.5", "context_length": 1048576, "pricing": { "prompt": "0.000000168", "completion": "0.000000336", "input_cache_read": "0.0000000034", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.93556701030928, "uptime_last_5m": 99.47780678851174, "uptime_last_1d": 98.68536753463832, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "xiaomi/mimo-v2.5@deepinfra", "name": "Xiaomi: MiMo-V2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "xiaomi/mimo-v2.5", "canonicalSlug": "xiaomi/mimo-v2.5-20260422", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+audio+video->text", "input_modalities": [ "text", "audio", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...", "uptimeLast30m": 99.79061976549414, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | xiaomi/mimo-v2.5-20260422", "model_id": "xiaomi/mimo-v2.5", "model_name": "Xiaomi: MiMo-V2.5", "context_length": 262144, "pricing": { "prompt": "0.0000004", "completion": "0.000002", "input_cache_read": "0.00000008", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.79061976549414, "uptime_last_5m": 98.68421052631578, "uptime_last_1d": 97.06918722132228, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-image-2@openai", "name": "OpenAI: GPT-5.4 Image 2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000015" } ], "cacheRead": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-image-2", "canonicalSlug": "openai/gpt-5.4-image-2-20260421", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 272000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text+image", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "image", "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-image-2-20260421", "model_id": "openai/gpt-5.4-image-2", "model_name": "OpenAI: GPT-5.4 Image 2", "context_length": 272000, "pricing": { "prompt": "0.000008", "completion": "0.000015", "image_output": "0.00003", "web_search": "0.01", "input_cache_read": "0.000002", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "inclusionai/ling-2.6-flash@novita", "name": "inclusionAI: Ling-2.6-flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheRead": [ { "amount": 0.002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "inclusionai/ling-2.6-flash", "canonicalSlug": "inclusionai/ling-2.6-flash-20260421", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....", "uptimeLast30m": 99.73992938228481, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | inclusionai/ling-2.6-flash-20260421", "model_id": "inclusionai/ling-2.6-flash", "model_name": "inclusionAI: Ling-2.6-flash", "context_length": 262144, "pricing": { "prompt": "0.00000001", "completion": "0.00000003", "input_cache_read": "0.000000002", "discount": 0.9 }, "provider_name": "Novita", "tag": "novita", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.73992938228481, "uptime_last_5m": 100, "uptime_last_1d": 99.98563591730807, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "~anthropic/claude-opus-latest", "name": "Anthropic: Claude Opus Latest", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "canonicalSlug": "~anthropic/claude-opus-latest", "contextLength": 1000000, "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "include_reasoning", "max_completion_tokens", "max_tokens", "reasoning", "reasoning_effort", "response_format", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "verbosity" ], "description": "This model always redirects to the latest model in the Claude Opus family.", "endpointCount": 0 } }, { "id": "openrouter/pareto-code", "name": "Pareto Code Router", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": -1000000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "-1" } ], "output": [ { "amount": -1000000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "-1" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "canonicalSlug": "openrouter/pareto-code", "contextLength": 2000000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [], "description": "The Pareto Router maintains a tiered shortlist of strong coding models, ranked by [Artificial Analysis](https://artificialanalysis.ai/) coding percentiles. Set min_coding_score between 0 and 1 on the [pareto-router plugin](https://openrouter.ai/docs/guides/routing/routers/pareto-router#the-min_coding_score-parameter) to control how...", "endpointCount": 0 } }, { "id": "kwaivgi/kling-video-o1@atlascloud", "name": "Kling: Video O1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "kwaivgi/kling-video-o1", "canonicalSlug": "kwaivgi/kling-video-o1-20260420", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kling Video O1 is a video generation model from Kuaishou. It supports text and image inputs with video output, enabling text-to-video and image-to-video workflows. It is suited for cinematic content...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | kwaivgi/kling-video-o1-20260420", "model_id": "kwaivgi/kling-video-o1", "model_name": "Kling: Video O1", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/hailuo-2.3@minimax", "name": "MiniMax: Hailuo 2.3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/hailuo-2.3", "canonicalSlug": "minimax/hailuo-2.3-20260420", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Hailuo 2.3 is a video generation model from MiniMax. It accepts text prompts and reference images as input and generates video output, supporting both text-to-video and image-to-video workflows. It is...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/hailuo-2.3-20260420", "model_id": "minimax/hailuo-2.3", "model_name": "MiniMax: Hailuo 2.3", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@decart", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5684, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005684" } ], "output": [ { "amount": 3.332, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003332" } ], "cacheRead": [ { "amount": 0.0925, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000925" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Decart", "servingProviderSlug": "decart", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.43109987357775, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Decart | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.0000005684", "completion": "0.000003332", "input_cache_read": "0.0000000925", "discount": 0 }, "provider_name": "Decart", "tag": "decart/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.43109987357775, "uptime_last_5m": 99.581589958159, "uptime_last_1d": 95.94876026562929, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@chutes", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.58, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000058" } ], "output": [ { "amount": 3.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000034" } ], "cacheRead": [ { "amount": 0.058, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000058" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Chutes", "servingProviderSlug": "chutes", "contextLength": 262144, "maxCompletionTokens": 65535, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.60106382978722, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Chutes | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000058", "completion": "0.0000034", "input_cache_read": "0.000000058", "discount": 0 }, "provider_name": "Chutes", "tag": "chutes/int4", "quantization": "int4", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.60106382978722, "uptime_last_5m": 100, "uptime_last_1d": 96.51931084349366, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@streamlake", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5984999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005985" } ], "output": [ { "amount": 2.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000252" } ], "cacheRead": [ { "amount": 0.1008, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001008" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 256000, "maxCompletionTokens": 256000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "stop", "top_p", "temperature", "max_tokens", "tool_choice", "tools", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.37908496732027, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 256000, "pricing": { "prompt": "0.0000005985", "completion": "0.00000252", "input_cache_read": "0.0000001008", "discount": 0.37 }, "provider_name": "StreamLake", "tag": "streamlake/fp8", "quantization": "fp8", "max_completion_tokens": 256000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "stop", "top_p", "temperature", "max_tokens", "tool_choice", "tools", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.37908496732027, "uptime_last_5m": 99.09502262443439, "uptime_last_1d": 98.68770220655058, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@inceptron", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 3.41, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000341" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Inceptron", "servingProviderSlug": "inceptron", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "stop", "frequency_penalty", "logit_bias", "parallel_tool_calls", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.6262680192205, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Inceptron | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.0000006", "completion": "0.00000341", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Inceptron", "tag": "inceptron/int4", "quantization": "int4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "stop", "frequency_penalty", "logit_bias", "parallel_tool_calls", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.6262680192205, "uptime_last_5m": 100, "uptime_last_1d": 99.42636384921563, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@coreweave", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000065" } ], "output": [ { "amount": 3.41, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000341" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000065", "completion": "0.00000341", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 98.45192154124433, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@crusoe", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Crusoe", "servingProviderSlug": "crusoe", "contextLength": 262144, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.9244142101285, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Crusoe | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.0000007", "completion": "0.0000035", "input_cache_read": "0.00000035", "discount": 0 }, "provider_name": "Crusoe", "tag": "crusoe/bf16", "quantization": "bf16", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.9244142101285, "uptime_last_5m": 100, "uptime_last_1d": 99.1943817618154, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@parasail", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.78386167146974, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000075", "completion": "0.0000035", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/int4", "quantization": "int4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.78386167146974, "uptime_last_5m": 100, "uptime_last_1d": 99.67765415410211, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@venice", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 65536, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tool_choice", "tools", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 256000, "pricing": { "prompt": "0.00000075", "completion": "0.0000035", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "Venice", "tag": "venice/int4", "quantization": "int4", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tool_choice", "tools", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 99, "uptime_last_5m": null, "uptime_last_1d": 76.84319558058004, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@deepinfra", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "structured_outputs", "logit_bias" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.94266055045871, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000075", "completion": "0.0000035", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "structured_outputs", "logit_bias" ], "status": 0, "uptime_last_30m": 99.94266055045871, "uptime_last_5m": 100, "uptime_last_1d": 99.6638254494979, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@digitalocean", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.76, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000076" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000019" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "structured_outputs", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000076", "completion": "0.0000032", "input_cache_read": "0.00000019", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "structured_outputs", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.3489869065906, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@siliconflow", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.77, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000077" } ], "output": [ { "amount": 3.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000034" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "max_tokens" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000077", "completion": "0.0000034", "input_cache_read": "0.00000014", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "max_tokens" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.71657677063213, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@novita", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "output": [ { "amount": 3.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000034" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.88249118683902, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.0000008", "completion": "0.0000034", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "Novita", "tag": "novita", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.88249118683902, "uptime_last_5m": 100, "uptime_last_1d": 99.71476873032024, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@atlascloud", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "tool_choice", "tools", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.69650986342944, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/int4", "quantization": "int4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "tool_choice", "tools", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.69650986342944, "uptime_last_5m": 100, "uptime_last_1d": 97.09471237652527, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@moonshot-ai", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Moonshot AI", "servingProviderSlug": "moonshot-ai", "contextLength": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "frequency_penalty", "presence_penalty", "structured_outputs", "response_format", "tool_choice", "tools" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.919322307382, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Moonshot AI | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "Moonshot AI", "tag": "moonshotai/int4", "quantization": "int4", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "frequency_penalty", "presence_penalty", "structured_outputs", "response_format", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 99.919322307382, "uptime_last_5m": 100, "uptime_last_1d": 99.94173446161554, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@cloudflare", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "structured_outputs", "response_format", "min_p", "stop", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "structured_outputs", "response_format", "min_p", "stop", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96996915012694, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@baidu", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Baidu", "servingProviderSlug": "baidu", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "frequency_penalty", "presence_penalty", "stop", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.89299090422686, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Baidu | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "Baidu", "tag": "baidu/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "frequency_penalty", "presence_penalty", "stop", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.89299090422686, "uptime_last_5m": 100, "uptime_last_1d": 99.91875951537172, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@sail-research", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Sail Research", "servingProviderSlug": "sail-research", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "logprobs", "top_logprobs", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Sail Research | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.000001", "completion": "0.000004", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Sail Research", "tag": "sail-research/int4", "quantization": "int4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "stop", "seed", "max_tokens", "logprobs", "top_logprobs", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.86301369863013, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@phala", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.0899999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000109" } ], "output": [ { "amount": 4.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000046" } ], "cacheRead": [ { "amount": 0.37, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000037" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "structured_outputs", "response_format", "tool_choice", "tools" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 99.84411535463757, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000109", "completion": "0.0000046", "input_cache_read": "0.00000037", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "structured_outputs", "response_format", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 99.84411535463757, "uptime_last_5m": 99.58847736625515, "uptime_last_1d": 99.04553632025164, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@together", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 262144, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": 89.95918367346938, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.0000012", "completion": "0.0000045", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": -2, "uptime_last_30m": 89.95918367346938, "uptime_last_5m": 100, "uptime_last_1d": 80.65005958559182, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@baseten", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "BaseTen", "servingProviderSlug": "baseten", "contextLength": 262000, "maxCompletionTokens": 262000, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "BaseTen | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262000, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "BaseTen", "tag": "baseten/fp4", "quantization": "fp4", "max_completion_tokens": 262000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.6@fireworks", "name": "MoonshotAI: Kimi K2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.6", "canonicalSlug": "moonshotai/kimi-k2.6-20260420", "servingProvider": "Fireworks", "servingProviderSlug": "fireworks", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fireworks | moonshotai/kimi-k2.6-20260420", "model_id": "moonshotai/kimi-k2.6", "model_name": "MoonshotAI: Kimi K2.6", "context_length": 262144, "pricing": { "prompt": "0.00000095", "completion": "0.000004", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "Fireworks", "tag": "fireworks", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 0, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/voxtral-mini-tts-2603@mistral", "name": "Mistral: Voxtral Mini TTS", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000016" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/voxtral-mini-tts-2603", "canonicalSlug": "mistralai/voxtral-mini-tts-2603", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs" ], "architecture": { "modality": "text->speech", "input_modalities": [ "text" ], "output_modalities": [ "speech" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Voxtral Mini TTS is Mistral's text-to-speech model featuring zero-shot voice cloning and multilingual support. It converts text input into natural-sounding audio output.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/voxtral-mini-tts-2603", "model_id": "mistralai/voxtral-mini-tts-2603", "model_name": "Mistral: Voxtral Mini TTS", "context_length": 4096, "pricing": { "prompt": "0.000016", "completion": "0", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-embedding-2-preview@google-ai-studio", "name": "Google: Gemini Embedding 2 Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.0000002" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-embedding-2-preview", "canonicalSlug": "google/gemini-embedding-2-preview", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text+image+file+audio+video->embeddings", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "embeddings" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini Embedding 2 Preview is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-embedding-2-preview", "model_id": "google/gemini-embedding-2-preview", "model_name": "Google: Gemini Embedding 2 Preview", "context_length": 8192, "pricing": { "prompt": "0.0000002", "completion": "0", "image": "0.00000045", "audio": "0.0000065", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.7@amazon-bedrock", "name": "Anthropic: Claude Opus 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.7", "canonicalSlug": "anthropic/claude-4.7-opus-20260416", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...", "uptimeLast30m": 99.80806142034548, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.7-opus-20260416", "model_id": "anthropic/claude-opus-4.7", "model_name": "Anthropic: Claude Opus 4.7", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.80806142034548, "uptime_last_5m": 100, "uptime_last_1d": 99.92653136192487, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.7@google", "name": "Anthropic: Claude Opus 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.7", "canonicalSlug": "anthropic/claude-4.7-opus-20260416", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.7-opus-20260416", "model_id": "anthropic/claude-opus-4.7", "model_name": "Anthropic: Claude Opus 4.7", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9782324771441, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.7@anthropic", "name": "Anthropic: Claude Opus 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.7", "canonicalSlug": "anthropic/claude-4.7-opus-20260416", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.7-opus-20260416", "model_id": "anthropic/claude-opus-4.7", "model_name": "Anthropic: Claude Opus 4.7", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.89815836374439, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.7@claude-platform-on-aws", "name": "Anthropic: Claude Opus 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.7", "canonicalSlug": "anthropic/claude-4.7-opus-20260416", "servingProvider": "Claude Platform on AWS", "servingProviderSlug": "claude-platform-on-aws", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Claude Platform on AWS | anthropic/claude-4.7-opus-20260416", "model_id": "anthropic/claude-opus-4.7", "model_name": "Anthropic: Claude Opus 4.7", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Claude Platform on AWS", "tag": "claude-on-aws", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.77976624007215, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.7@azure", "name": "Anthropic: Claude Opus 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.7", "canonicalSlug": "anthropic/claude-4.7-opus-20260416", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-4.7-opus-20260416", "model_id": "anthropic/claude-opus-4.7", "model_name": "Anthropic: Claude Opus 4.7", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Azure", "tag": "azure/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.7@amazon-bedrock", "name": "Anthropic: Claude Opus 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.7", "canonicalSlug": "anthropic/claude-4.7-opus-20260416", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.7-opus-20260416", "model_id": "anthropic/claude-opus-4.7", "model_name": "Anthropic: Claude Opus 4.7", "context_length": 1000000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.7@google", "name": "Anthropic: Claude Opus 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.7", "canonicalSlug": "anthropic/claude-4.7-opus-20260416", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.7-opus-20260416", "model_id": "anthropic/claude-opus-4.7", "model_name": "Anthropic: Claude Opus 4.7", "context_length": 1000000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/europe", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.7@google", "name": "Anthropic: Claude Opus 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.7", "canonicalSlug": "anthropic/claude-4.7-opus-20260416", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.7-opus-20260416", "model_id": "anthropic/claude-opus-4.7", "model_name": "Anthropic: Claude Opus 4.7", "context_length": 1000000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.7:batch@anthropic", "name": "Anthropic: Claude Opus 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [ { "amount": 3.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.7:batch", "canonicalSlug": "anthropic/claude-4.7-opus-20260416", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.7-opus-20260416:batch", "model_id": "anthropic/claude-opus-4.7:batch", "model_name": "Anthropic: Claude Opus 4.7", "context_length": 1000000, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "input_cache_write_1h": "0.000005", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "alibaba/wan-2.7@atlascloud", "name": "Alibaba: Wan 2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "alibaba/wan-2.7", "canonicalSlug": "alibaba/wan-2.7-20260414", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Wan 2.7 is a video generation model from Alibaba. It supports text-to-video, image-to-video with first and last frame control, and reference-to-video, where multiple reference images guide the style and content...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | alibaba/wan-2.7-20260414", "model_id": "alibaba/wan-2.7", "model_name": "Alibaba: Wan 2.7", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance/seedance-2.0@seed", "name": "ByteDance: Seedance 2.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance/seedance-2.0", "canonicalSlug": "bytedance/seedance-2.0-20260414", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "frequency_penalty" ], "architecture": { "modality": "text+image+audio+video->video", "input_modalities": [ "text", "image", "video", "audio" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Seedance 2.0 is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It is particularly strong at preserving character consistency,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance/seedance-2.0-20260414", "model_id": "bytedance/seedance-2.0", "model_name": "ByteDance: Seedance 2.0", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Seed", "tag": "seed", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "frequency_penalty" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance/seedance-2.0-fast@seed", "name": "ByteDance: Seedance 2.0 Fast", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance/seedance-2.0-fast", "canonicalSlug": "bytedance/seedance-2.0-fast-20260414", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "frequency_penalty" ], "architecture": { "modality": "text+image+audio+video->video", "input_modalities": [ "text", "image", "video", "audio" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Seedance 2.0 Fast is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It prioritizes generation speed and lower cost...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance/seedance-2.0-fast-20260414", "model_id": "bytedance/seedance-2.0-fast", "model_name": "ByteDance: Seedance 2.0 Fast", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Seed", "tag": "seed", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "frequency_penalty" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@gmicloud", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.9099999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000091" } ], "output": [ { "amount": 2.8600000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000286" } ], "cacheRead": [ { "amount": 0.16899999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000169" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 202752, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 99.50124688279301, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202752, "pricing": { "prompt": "0.00000091", "completion": "0.00000286", "input_cache_read": "0.000000169", "discount": 0.35 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.50124688279301, "uptime_last_5m": 100, "uptime_last_1d": 99.38035097677499, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@streamlake", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.966, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000966" } ], "output": [ { "amount": 3.036, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003036" } ], "cacheRead": [ { "amount": 0.1794, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001794" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 200000, "maxCompletionTokens": 128000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 99.88700564971752, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 200000, "pricing": { "prompt": "0.000000966", "completion": "0.000003036", "input_cache_read": "0.0000001794", "discount": 0.31 }, "provider_name": "StreamLake", "tag": "streamlake/fp8", "quantization": "fp8", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop" ], "status": 0, "uptime_last_30m": 99.88700564971752, "uptime_last_5m": 100, "uptime_last_1d": 98.3624600282591, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@digitalocean", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.975, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000975" } ], "output": [ { "amount": 4.300000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000043" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 163840, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 92.831541218638, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 163840, "pricing": { "prompt": "0.000000975", "completion": "0.0000043", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": -2, "uptime_last_30m": 92.831541218638, "uptime_last_5m": 92.10526315789474, "uptime_last_1d": 91.91451990632319, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@chutes", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.98, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000098" } ], "output": [ { "amount": 3.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000308" } ], "cacheRead": [ { "amount": 0.098, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000098" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "Chutes", "servingProviderSlug": "chutes", "contextLength": 202752, "maxCompletionTokens": 65535, "quantization": "fp8", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "structured_outputs", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 86.58008658008657, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Chutes | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202752, "pricing": { "prompt": "0.00000098", "completion": "0.00000308", "input_cache_read": "0.000000098", "discount": 0 }, "provider_name": "Chutes", "tag": "chutes/fp8", "quantization": "fp8", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "structured_outputs", "tools", "tool_choice", "response_format" ], "status": -2, "uptime_last_30m": 86.58008658008657, "uptime_last_5m": 88.07339449541286, "uptime_last_1d": 71.9188767550702, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@deepinfra", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.0499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000105" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.205, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000205" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 202752, "maxCompletionTokens": 65536, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202752, "pricing": { "prompt": "0.00000105", "completion": "0.0000035", "input_cache_read": "0.000000205", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 98.56046065259117, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@siliconflow", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000119" } ], "output": [ { "amount": 3.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000374" } ], "cacheRead": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 204800, "pricing": { "prompt": "0.00000119", "completion": "0.00000374", "input_cache_read": "0.0000006", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.87760709379293, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@crusoe", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "Crusoe", "servingProviderSlug": "crusoe", "contextLength": 202752, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 98.94016339147714, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Crusoe | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202752, "pricing": { "prompt": "0.0000012", "completion": "0.0000044", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "Crusoe", "tag": "crusoe/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 98.94016339147714, "uptime_last_5m": 98.65125240847784, "uptime_last_1d": 99.63855178401954, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@phala", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000121" } ], "output": [ { "amount": 4.199999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000042" } ], "cacheRead": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 202752, "maxCompletionTokens": 128000, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "tool_choice", "tools", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 89.01273885350318, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202752, "pricing": { "prompt": "0.00000121", "completion": "0.0000042", "input_cache_read": "0.0000006", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "tool_choice", "tools", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": -2, "uptime_last_30m": 89.01273885350318, "uptime_last_5m": 90.54054054054053, "uptime_last_1d": 71.67891464104014, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@atlascloud", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000126" } ], "output": [ { "amount": 3.9600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000396" } ], "cacheRead": [ { "amount": 0.234, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000234" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 202752, "maxCompletionTokens": 202752, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 99.58706125258087, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202752, "pricing": { "prompt": "0.00000126", "completion": "0.00000396", "input_cache_read": "0.000000234", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 202752, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.58706125258087, "uptime_last_5m": 99.26470588235294, "uptime_last_1d": 98.41952798841271, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@alibaba", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000133" } ], "output": [ { "amount": 4.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000418" } ], "cacheRead": [ { "amount": 0.24699999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000247" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 202745, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "structured_outputs", "tools", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202745, "pricing": { "prompt": "0.00000133", "completion": "0.00000418", "input_cache_read": "0.000000247", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": 169984, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "structured_outputs", "tools", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.87725844461902, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@novita", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 204800, "pricing": { "prompt": "0.00000138", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.79641693811075, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@nebius", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 202752, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202752, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 95.37596334097063, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@parasail", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 99.85835694050992, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202752, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.85835694050992, "uptime_last_5m": 100, "uptime_last_1d": 98.48860257680873, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@friendli", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "Friendli", "servingProviderSlug": "friendli", "contextLength": 202752, "maxCompletionTokens": 202752, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Friendli | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202752, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Friendli", "tag": "friendli", "quantization": "unknown", "max_completion_tokens": 202752, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.97560578955927, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@z.ai", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 99.08256880733946, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202752, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k", "response_format" ], "status": 0, "uptime_last_30m": 99.08256880733946, "uptime_last_5m": 98.46153846153847, "uptime_last_1d": 99.03763953005544, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@baidu", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "Baidu", "servingProviderSlug": "baidu", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Baidu | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 202752, "pricing": { "prompt": "0.0000014", "completion": "0.0000044", "input_cache_read": "0.00000026", "discount": 0 }, "provider_name": "Baidu", "tag": "baidu/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.86162271726208, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5.1@venice", "name": "Z.ai: GLM 5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000154" } ], "output": [ { "amount": 4.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000484" } ], "cacheRead": [ { "amount": 0.286, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000286" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5.1", "canonicalSlug": "z-ai/glm-5.1-20260406", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 200000, "maxCompletionTokens": 80000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | z-ai/glm-5.1-20260406", "model_id": "z-ai/glm-5.1", "model_name": "Z.ai: GLM 5.1", "context_length": 200000, "pricing": { "prompt": "0.00000154", "completion": "0.00000484", "input_cache_read": "0.000000286", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 80000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 84.32967810399717, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "cohere/rerank-4-pro@cohere", "name": "Cohere: Rerank 4 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "cohere/rerank-4-pro", "canonicalSlug": "cohere/rerank-4-pro", "servingProvider": "Cohere", "servingProviderSlug": "cohere", "contextLength": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format" ], "architecture": { "modality": "text->rerank", "input_modalities": [ "text" ], "output_modalities": [ "rerank" ], "tokenizer": "Cohere", "instruct_type": null }, "description": "Cohere's AI search foundation model for enhancing the relevance of information surfaced within search and RAG systems. Features a 32K context window, multilingual support across 100+ languages, no data pre-processing...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cohere | cohere/rerank-4-pro", "model_id": "cohere/rerank-4-pro", "model_name": "Cohere: Rerank 4 Pro", "context_length": 32768, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Cohere", "tag": "cohere", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "cohere/rerank-4-fast@cohere", "name": "Cohere: Rerank 4 Fast", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "cohere/rerank-4-fast", "canonicalSlug": "cohere/rerank-4-fast", "servingProvider": "Cohere", "servingProviderSlug": "cohere", "contextLength": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format" ], "architecture": { "modality": "text->rerank", "input_modalities": [ "text" ], "output_modalities": [ "rerank" ], "tokenizer": "Cohere", "instruct_type": null }, "description": "Cohere's AI search foundation model for enhancing the relevance of information surfaced within search and RAG systems. Features a 32K context window, multilingual support across 100+ languages, no data pre-processing...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cohere | cohere/rerank-4-fast", "model_id": "cohere/rerank-4-fast", "model_name": "Cohere: Rerank 4 Fast", "context_length": 32768, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Cohere", "tag": "cohere", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "cohere/rerank-v3.5@cohere", "name": "Cohere: Rerank v3.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "cohere/rerank-v3.5", "canonicalSlug": "cohere/rerank-v3.5", "servingProvider": "Cohere", "servingProviderSlug": "cohere", "contextLength": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format" ], "architecture": { "modality": "text->rerank", "input_modalities": [ "text" ], "output_modalities": [ "rerank" ], "tokenizer": "Cohere", "instruct_type": null }, "description": "Rerank v3.5 is designed to reorder search results for improved relevance. It supports multi-aspect and semi-structured data reranking over 100+ languages. Ideal for refining results from semantic or keyword search...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cohere | cohere/rerank-v3.5", "model_id": "cohere/rerank-v3.5", "model_name": "Cohere: Rerank v3.5", "context_length": 4096, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Cohere", "tag": "cohere", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-26b-a4b-it@deepinfra", "name": "Google: Gemma 4 26B A4B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "output": [ { "amount": 0.33999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000034" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-26b-a4b-it", "canonicalSlug": "google/gemma-4-26b-a4b-it-20260403", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "structured_outputs", "logit_bias" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "uptimeLast30m": 99.8678897716912, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | google/gemma-4-26b-a4b-it-20260403", "model_id": "google/gemma-4-26b-a4b-it", "model_name": "Google: Gemma 4 26B A4B ", "context_length": 262144, "pricing": { "prompt": "0.00000007", "completion": "0.00000034", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "structured_outputs", "logit_bias" ], "status": 0, "uptime_last_30m": 99.8678897716912, "uptime_last_5m": 99.68340768455893, "uptime_last_1d": 99.71925980026543, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-26b-a4b-it@cloudflare", "name": "Google: Gemma 4 26B A4B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-26b-a4b-it", "canonicalSlug": "google/gemma-4-26b-a4b-it-20260403", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 256000, "maxCompletionTokens": 256000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "uptimeLast30m": 99.9670643343336, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | google/gemma-4-26b-a4b-it-20260403", "model_id": "google/gemma-4-26b-a4b-it", "model_name": "Google: Gemma 4 26B A4B ", "context_length": 256000, "pricing": { "prompt": "0.0000001", "completion": "0.0000003", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 256000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.9670643343336, "uptime_last_5m": 100, "uptime_last_1d": 99.84204087753714, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-26b-a4b-it@nextbit", "name": "Google: Gemma 4 26B A4B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-26b-a4b-it", "canonicalSlug": "google/gemma-4-26b-a4b-it-20260403", "servingProvider": "NextBit", "servingProviderSlug": "nextbit", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "repetition_penalty", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "uptimeLast30m": 99.87797272012365, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "NextBit | google/gemma-4-26b-a4b-it-20260403", "model_id": "google/gemma-4-26b-a4b-it", "model_name": "Google: Gemma 4 26B A4B ", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.0000004", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "NextBit", "tag": "nextbit/bf16", "quantization": "bf16", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "repetition_penalty", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.87797272012365, "uptime_last_5m": 99.9405705229794, "uptime_last_1d": 99.15680787064863, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-26b-a4b-it@siliconflow", "name": "Google: Gemma 4 26B A4B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-26b-a4b-it", "canonicalSlug": "google/gemma-4-26b-a4b-it-20260403", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "uptimeLast30m": 99.75354282193469, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | google/gemma-4-26b-a4b-it-20260403", "model_id": "google/gemma-4-26b-a4b-it", "model_name": "Google: Gemma 4 26B A4B ", "context_length": 262144, "pricing": { "prompt": "0.00000012", "completion": "0.0000004", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.75354282193469, "uptime_last_5m": 100, "uptime_last_1d": 96.39668277023306, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-26b-a4b-it@novita", "name": "Google: Gemma 4 26B A4B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-26b-a4b-it", "canonicalSlug": "google/gemma-4-26b-a4b-it-20260403", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "uptimeLast30m": 99.87351112048066, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | google/gemma-4-26b-a4b-it-20260403", "model_id": "google/gemma-4-26b-a4b-it", "model_name": "Google: Gemma 4 26B A4B ", "context_length": 262144, "pricing": { "prompt": "0.00000013", "completion": "0.0000004", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.87351112048066, "uptime_last_5m": 99.34264585045193, "uptime_last_1d": 94.5577167616446, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-26b-a4b-it@parasail", "name": "Google: Gemma 4 26B A4B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-26b-a4b-it", "canonicalSlug": "google/gemma-4-26b-a4b-it-20260403", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "uptimeLast30m": 99.60647936304566, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | google/gemma-4-26b-a4b-it-20260403", "model_id": "google/gemma-4-26b-a4b-it", "model_name": "Google: Gemma 4 26B A4B ", "context_length": 262144, "pricing": { "prompt": "0.00000013", "completion": "0.0000004", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.60647936304566, "uptime_last_5m": 99.71978984238179, "uptime_last_1d": 98.89477200363885, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-26b-a4b-it@venice", "name": "Google: Gemma 4 26B A4B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-26b-a4b-it", "canonicalSlug": "google/gemma-4-26b-a4b-it-20260403", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 8192, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "uptimeLast30m": 99.69244685662596, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | google/gemma-4-26b-a4b-it-20260403", "model_id": "google/gemma-4-26b-a4b-it", "model_name": "Google: Gemma 4 26B A4B ", "context_length": 256000, "pricing": { "prompt": "0.00000013", "completion": "0.0000004", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "Venice", "tag": "venice/bf16", "quantization": "bf16", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.69244685662596, "uptime_last_5m": 99.71929824561403, "uptime_last_1d": 99.35966161396796, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-26b-a4b-it@google", "name": "Google: Gemma 4 26B A4B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-26b-a4b-it", "canonicalSlug": "google/gemma-4-26b-a4b-it-20260403", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "uptimeLast30m": 99.5141588006663, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemma-4-26b-a4b-it-20260403", "model_id": "google/gemma-4-26b-a4b-it", "model_name": "Google: Gemma 4 26B A4B ", "context_length": 262144, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.5141588006663, "uptime_last_5m": 100, "uptime_last_1d": 97.04274391304169, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-26b-a4b-it:free@google-ai-studio", "name": "Google: Gemma 4 26B A4B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-26b-a4b-it:free", "canonicalSlug": "google/gemma-4-26b-a4b-it-20260403", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemma-4-26b-a4b-it-20260403:free", "model_id": "google/gemma-4-26b-a4b-it:free", "model_name": "Google: Gemma 4 26B A4B ", "context_length": 262144, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.52818677296023, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-26b-a4b-it:free@darkbloom", "name": "Google: Gemma 4 26B A4B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-26b-a4b-it:free", "canonicalSlug": "google/gemma-4-26b-a4b-it-20260403", "servingProvider": "Darkbloom", "servingProviderSlug": "darkbloom", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...", "uptimeLast30m": 97.29383096062588, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Darkbloom | google/gemma-4-26b-a4b-it-20260403:free", "model_id": "google/gemma-4-26b-a4b-it:free", "model_name": "Google: Gemma 4 26B A4B ", "context_length": 131072, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Darkbloom", "tag": "darkbloom", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 97.29383096062588, "uptime_last_5m": 97.59358288770053, "uptime_last_1d": 95.62850368280935, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@openinference", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "OpenInference", "servingProviderSlug": "openinference", "contextLength": 262144, "maxCompletionTokens": 8192, "quantization": "bf16", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "stop", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "top_p", "frequency_penalty", "presence_penalty" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 94.67502649240551, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenInference | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000008", "completion": "0.00000035", "input_cache_read": "0.00000001", "discount": 0 }, "provider_name": "OpenInference", "tag": "open-inference/bf16", "quantization": "bf16", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "stop", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "top_p", "frequency_penalty", "presence_penalty" ], "status": -2, "uptime_last_30m": 94.67502649240551, "uptime_last_5m": 90.41353383458647, "uptime_last_1d": 94.86571893719824, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@deepinfra", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "output": [ { "amount": 0.33999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000034" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 99.87483092080028, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000009", "completion": "0.00000034", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/turbo", "quantization": "fp4", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.87483092080028, "uptime_last_5m": 99.9844527363184, "uptime_last_1d": 99.85005316091107, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@coreweave", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.33999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000034" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 99.13281606572342, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.00000034", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/bf16", "quantization": "bf16", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.13281606572342, "uptime_last_5m": 99.00317172632532, "uptime_last_1d": 97.68068719377355, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@venice", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "output": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000036" } ], "cacheRead": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 8192, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 99.89431409849927, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 256000, "pricing": { "prompt": "0.00000012", "completion": "0.00000036", "input_cache_read": "0.00000009", "discount": 0 }, "provider_name": "Venice", "tag": "venice/bf16", "quantization": "bf16", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.89431409849927, "uptime_last_5m": 99.53426480372588, "uptime_last_1d": 99.77506134458332, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@chutes", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "output": [ { "amount": 0.37, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000037" } ], "cacheRead": [ { "amount": 0.012, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "Chutes", "servingProviderSlug": "chutes", "contextLength": 131072, "maxCompletionTokens": 65536, "quantization": "fp4", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 93.63112391930835, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Chutes | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 131072, "pricing": { "prompt": "0.00000012", "completion": "0.00000037", "input_cache_read": "0.000000012", "discount": 0 }, "provider_name": "Chutes", "tag": "chutes/fp4", "quantization": "fp4", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "tools", "tool_choice", "structured_outputs" ], "status": -2, "uptime_last_30m": 93.63112391930835, "uptime_last_5m": 76.30522088353415, "uptime_last_1d": 87.02337694742916, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@deepinfra", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000038" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp8", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "structured_outputs", "logit_bias" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 91.95646402973188, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000013", "completion": "0.00000038", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "structured_outputs", "logit_bias" ], "status": -2, "uptime_last_30m": 91.95646402973188, "uptime_last_5m": 99.65277777777779, "uptime_last_1d": 98.42866277715265, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@siliconflow", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "response_format", "structured_outputs", "tool_choice", "tools" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 97.52303923474744, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000013", "completion": "0.0000004", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "response_format", "structured_outputs", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 97.52303923474744, "uptime_last_5m": 98.34555367550291, "uptime_last_1d": 94.51526526072739, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@novita", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "bf16", "status": -5, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 68.94980420078319, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000014", "completion": "0.0000004", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs", "structured_outputs" ], "status": -5, "uptime_last_30m": 68.94980420078319, "uptime_last_5m": 85.04273504273505, "uptime_last_1d": 81.39202821141566, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@friendli", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "Friendli", "servingProviderSlug": "friendli", "contextLength": 262144, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 99.98951056286319, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Friendli | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000014", "completion": "0.0000004", "discount": 0 }, "provider_name": "Friendli", "tag": "friendli", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.98951056286319, "uptime_last_5m": 100, "uptime_last_1d": 99.46138214356444, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@crusoe", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "Crusoe", "servingProviderSlug": "crusoe", "contextLength": 262144, "maxCompletionTokens": 262141, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 98.19477434679335, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Crusoe | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000014", "completion": "0.0000004", "input_cache_read": "0.00000014", "discount": 0 }, "provider_name": "Crusoe", "tag": "crusoe", "quantization": "unknown", "max_completion_tokens": 262141, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 98.19477434679335, "uptime_last_5m": 99.51690821256038, "uptime_last_1d": 99.03396717930627, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@parasail", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 99.52780692549842, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000015", "completion": "0.0000004", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.52780692549842, "uptime_last_5m": 99.28418038654259, "uptime_last_1d": 99.22663099332273, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@phala", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.45999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000046" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 99.29824561403508, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000015", "completion": "0.00000046", "input_cache_read": "0.000000075", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.29824561403508, "uptime_last_5m": null, "uptime_last_1d": 98.75480088942794, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@deepinfra", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 0.76, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000076" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 8192, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 98.63481228668942, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 131072, "pricing": { "prompt": "0.00000027", "completion": "0.00000076", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/ultra", "quantization": "fp8", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 98.63481228668942, "uptime_last_5m": 100, "uptime_last_1d": 83.54356458771315, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@together", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "output": [ { "amount": 0.86, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000086" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 262144, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 80.76923076923077, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000028", "completion": "0.00000086", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format" ], "status": -2, "uptime_last_30m": 80.76923076923077, "uptime_last_5m": null, "uptime_last_1d": 83.4039206195547, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@sambanova", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000038" } ], "output": [ { "amount": 1.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000115" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "SambaNova", "servingProviderSlug": "sambanova", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SambaNova | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 131072, "pricing": { "prompt": "0.00000038", "completion": "0.00000115", "discount": 0 }, "provider_name": "SambaNova", "tag": "sambanova", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.20333212125219, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@together", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000039" } ], "output": [ { "amount": 0.9700000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000097" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "structured_outputs", "response_format", "tool_choice", "tools" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 99.22330097087378, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000039", "completion": "0.00000097", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "structured_outputs", "response_format", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 99.22330097087378, "uptime_last_5m": 100, "uptime_last_1d": 91.23232661256588, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@modelrun", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "ModelRun", "servingProviderSlug": "modelrun", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "presence_penalty", "repetition_penalty", "frequency_penalty", "top_p", "top_k", "stop", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 99.94459322190414, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "ModelRun | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0.00000075", "completion": "0.000001", "input_cache_read": "0.00000075", "discount": 0 }, "provider_name": "ModelRun", "tag": "modelrun/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "presence_penalty", "repetition_penalty", "frequency_penalty", "top_p", "top_k", "stop", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.94459322190414, "uptime_last_5m": 99.93548387096774, "uptime_last_1d": 99.92027392588702, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it@cerebras", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.9900000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000099" } ], "output": [ { "amount": 1.49, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000149" } ], "cacheRead": [ { "amount": 0.9900000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000099" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "Cerebras", "servingProviderSlug": "cerebras", "contextLength": 131072, "maxCompletionTokens": 40960, "quantization": "fp16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "seed", "tools", "tool_choice", "frequency_penalty", "presence_penalty", "logit_bias" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cerebras | google/gemma-4-31b-it-20260402", "model_id": "google/gemma-4-31b-it", "model_name": "Google: Gemma 4 31B", "context_length": 131072, "pricing": { "prompt": "0.00000099", "completion": "0.00000149", "input_cache_read": "0.00000099", "discount": 0 }, "provider_name": "Cerebras", "tag": "cerebras/fp16", "quantization": "fp16", "max_completion_tokens": 40960, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "seed", "tools", "tool_choice", "frequency_penalty", "presence_penalty", "logit_bias" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99394372967836, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-4-31b-it:free@google-ai-studio", "name": "Google: Gemma 4 31B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-4-31b-it:free", "canonicalSlug": "google/gemma-4-31b-it-20260402", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemma", "instruct_type": null }, "description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemma-4-31b-it-20260402:free", "model_id": "google/gemma-4-31b-it:free", "model_name": "Google: Gemma 4 31B", "context_length": 262144, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.76815434296597, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.6-plus@alibaba", "name": "Qwen: Qwen3.6 Plus", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.325, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000325" } ], "output": [ { "amount": 1.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000195" } ], "cacheRead": [], "cacheWrite": [ { "amount": 0.40625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000040625" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.6-plus", "canonicalSlug": "qwen/qwen3.6-plus-04-02", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.6-plus-04-02", "model_id": "qwen/qwen3.6-plus", "model_name": "Qwen: Qwen3.6 Plus", "context_length": 1000000, "pricing": { "prompt": "0.000000325", "completion": "0.00000195", "input_cache_write": "0.00000040625", "discount": 0, "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.0000013", "completion": "0.0000039", "input_cache_write": "0.000001625" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 983616, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99932790881067, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5v-turbo@z.ai", "name": "Z.ai: GLM 5V Turbo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000024" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5v-turbo", "canonicalSlug": "z-ai/glm-5v-turbo-20260401", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "top_k" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...", "uptimeLast30m": 98.4540276647681, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-5v-turbo-20260401", "model_id": "z-ai/glm-5v-turbo", "model_name": "Z.ai: GLM 5V Turbo", "context_length": 202752, "pricing": { "prompt": "0.0000012", "completion": "0.000004", "input_cache_read": "0.00000024", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "top_k" ], "status": 0, "uptime_last_30m": 98.4540276647681, "uptime_last_5m": 96.37681159420289, "uptime_last_1d": 98.6306295508345, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "arcee-ai/trinity-large-thinking@parasail", "name": "Arcee AI: Trinity Large Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000085" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "arcee-ai/trinity-large-thinking", "canonicalSlug": "arcee-ai/trinity-large-thinking", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | arcee-ai/trinity-large-thinking", "model_id": "arcee-ai/trinity-large-thinking", "model_name": "Arcee AI: Trinity Large Thinking", "context_length": 262144, "pricing": { "prompt": "0.00000022", "completion": "0.00000085", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp4", "quantization": "fp4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99280316660669, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "arcee-ai/trinity-large-thinking@arcee-ai", "name": "Arcee AI: Trinity Large Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "arcee-ai/trinity-large-thinking", "canonicalSlug": "arcee-ai/trinity-large-thinking", "servingProvider": "Arcee AI", "servingProviderSlug": "arcee-ai", "contextLength": 262144, "maxCompletionTokens": 80000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_k", "top_p", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Arcee AI | arcee-ai/trinity-large-thinking", "model_id": "arcee-ai/trinity-large-thinking", "model_name": "Arcee AI: Trinity Large Thinking", "context_length": 262144, "pricing": { "prompt": "0.00000025", "completion": "0.0000008", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Arcee AI", "tag": "arcee-ai", "quantization": "unknown", "max_completion_tokens": 80000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_k", "top_p", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.20-multi-agent@xai", "name": "SpaceXAI: Grok 4.20 Multi-Agent", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.20-multi-agent", "canonicalSlug": "x-ai/grok-4.20-multi-agent-20260309", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 2000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.20-multi-agent-20260309", "model_id": "x-ai/grok-4.20-multi-agent", "model_name": "SpaceXAI: Grok 4.20 Multi-Agent", "context_length": 2000000, "pricing": { "prompt": "0.00000125", "completion": "0.0000025", "web_search": "0.005", "input_cache_read": "0.0000002", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000005", "input_cache_read": "0.0000004" } ] }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 70.38292104371399, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.20-multi-agent@xai", "name": "SpaceXAI: Grok 4.20 Multi-Agent", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.20-multi-agent", "canonicalSlug": "x-ai/grok-4.20-multi-agent-20260309", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 2000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.20-multi-agent-20260309", "model_id": "x-ai/grok-4.20-multi-agent", "model_name": "SpaceXAI: Grok 4.20 Multi-Agent", "context_length": 2000000, "pricing": { "prompt": "0.0000025", "completion": "0.000005", "web_search": "0.005", "input_cache_read": "0.0000004", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000005", "completion": "0.00001", "input_cache_read": "0.0000008" } ] }, "provider_name": "xAI", "tag": "xai/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 70.38292104371399, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.20-multi-agent@xai", "name": "SpaceXAI: Grok 4.20 Multi-Agent", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.20-multi-agent", "canonicalSlug": "x-ai/grok-4.20-multi-agent-20260309", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 2000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.20-multi-agent-20260309", "model_id": "x-ai/grok-4.20-multi-agent", "model_name": "SpaceXAI: Grok 4.20 Multi-Agent", "context_length": 2000000, "pricing": { "prompt": "0.00000125", "completion": "0.0000025", "web_search": "0.005", "input_cache_read": "0.0000002", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000005", "input_cache_read": "0.0000004" } ] }, "provider_name": "xAI", "tag": "xai/zdr", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 75.67164179104478, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.20-multi-agent@xai", "name": "SpaceXAI: Grok 4.20 Multi-Agent", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.20-multi-agent", "canonicalSlug": "x-ai/grok-4.20-multi-agent-20260309", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 2000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.20-multi-agent-20260309", "model_id": "x-ai/grok-4.20-multi-agent", "model_name": "SpaceXAI: Grok 4.20 Multi-Agent", "context_length": 2000000, "pricing": { "prompt": "0.0000025", "completion": "0.000005", "web_search": "0.005", "input_cache_read": "0.0000004", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000005", "completion": "0.00001", "input_cache_read": "0.0000008" } ] }, "provider_name": "xAI", "tag": "xai/zdr/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 75.67164179104478, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.20@xai", "name": "SpaceXAI: Grok 4.20", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.20", "canonicalSlug": "x-ai/grok-4.20-20260309", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 2000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...", "uptimeLast30m": 99.95610184372256, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.20-20260309", "model_id": "x-ai/grok-4.20", "model_name": "SpaceXAI: Grok 4.20", "context_length": 2000000, "pricing": { "prompt": "0.00000125", "completion": "0.0000025", "web_search": "0.005", "input_cache_read": "0.0000002", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000005", "input_cache_read": "0.0000004" } ] }, "provider_name": "xAI", "tag": "xai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.95610184372256, "uptime_last_5m": 100, "uptime_last_1d": 99.83489682499835, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.20@xai", "name": "SpaceXAI: Grok 4.20", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.20", "canonicalSlug": "x-ai/grok-4.20-20260309", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 2000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...", "uptimeLast30m": 99.95610184372256, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.20-20260309", "model_id": "x-ai/grok-4.20", "model_name": "SpaceXAI: Grok 4.20", "context_length": 2000000, "pricing": { "prompt": "0.0000025", "completion": "0.000005", "web_search": "0.005", "input_cache_read": "0.0000004", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000005", "completion": "0.00001", "input_cache_read": "0.0000008" } ] }, "provider_name": "xAI", "tag": "xai/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.95610184372256, "uptime_last_5m": 100, "uptime_last_1d": 99.83489682499835, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.20@xai", "name": "SpaceXAI: Grok 4.20", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.20", "canonicalSlug": "x-ai/grok-4.20-20260309", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 2000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...", "uptimeLast30m": 99.90089197224975, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.20-20260309", "model_id": "x-ai/grok-4.20", "model_name": "SpaceXAI: Grok 4.20", "context_length": 2000000, "pricing": { "prompt": "0.00000125", "completion": "0.0000025", "web_search": "0.005", "input_cache_read": "0.0000002", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000005", "input_cache_read": "0.0000004" } ] }, "provider_name": "xAI", "tag": "xai/zdr", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.90089197224975, "uptime_last_5m": 100, "uptime_last_1d": 99.74417542256738, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "x-ai/grok-4.20@xai", "name": "SpaceXAI: Grok 4.20", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "x-ai/grok-4.20", "canonicalSlug": "x-ai/grok-4.20-20260309", "servingProvider": "xAI", "servingProviderSlug": "xai", "contextLength": 2000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Grok", "instruct_type": null }, "description": "Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...", "uptimeLast30m": 99.90089197224975, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "xAI | x-ai/grok-4.20-20260309", "model_id": "x-ai/grok-4.20", "model_name": "SpaceXAI: Grok 4.20", "context_length": 2000000, "pricing": { "prompt": "0.0000025", "completion": "0.000005", "web_search": "0.005", "input_cache_read": "0.0000004", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000005", "completion": "0.00001", "input_cache_read": "0.0000008" } ] }, "provider_name": "xAI", "tag": "xai/zdr/priority", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "response_format", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.90089197224975, "uptime_last_5m": 100, "uptime_last_1d": 99.74417542256738, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/lyria-3-pro-preview@google-ai-studio", "name": "Google: Lyria 3 Pro Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/lyria-3-pro-preview", "canonicalSlug": "google/lyria-3-pro-preview-20260330", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text+image->text+audio", "input_modalities": [ "text", "image" ], "output_modalities": [ "text", "audio" ], "tokenizer": "Other", "instruct_type": null }, "description": "Full-length songs are priced at $0.08 per song. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate high-quality, 48kHz...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/lyria-3-pro-preview-20260330", "model_id": "google/lyria-3-pro-preview", "model_name": "Google: Lyria 3 Pro Preview", "context_length": 1048576, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.95032290114257, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/lyria-3-clip-preview@google-ai-studio", "name": "Google: Lyria 3 Clip Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/lyria-3-clip-preview", "canonicalSlug": "google/lyria-3-clip-preview-20260330", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text+image->text+audio", "input_modalities": [ "text", "image" ], "output_modalities": [ "text", "audio" ], "tokenizer": "Other", "instruct_type": null }, "description": "30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/lyria-3-clip-preview-20260330", "model_id": "google/lyria-3-clip-preview", "model_name": "Google: Lyria 3 Clip Preview", "context_length": 1048576, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "alibaba/wan-2.6@atlascloud", "name": "Alibaba: Wan 2.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "alibaba/wan-2.6", "canonicalSlug": "alibaba/wan-2.6-20260327", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Alibaba's most advanced video generation model, supporting over 10 visual creation capabilities in a unified system. Wan 2.6 generates 1080p video at 24fps from text, images, reference videos, or audio,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | alibaba/wan-2.6-20260327", "model_id": "alibaba/wan-2.6", "model_name": "Alibaba: Wan 2.6", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "kwaipilot/kat-coder-pro-v2@streamlake", "name": "Kwaipilot: KAT-Coder-Pro V2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "kwaipilot/kat-coder-pro-v2", "canonicalSlug": "kwaipilot/kat-coder-pro-v2-20260327", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 256000, "maxCompletionTokens": 80000, "quantization": "unknown", "status": 0, "supportedParameters": [ "tool_choice", "tools", "max_tokens", "temperature", "top_p", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...", "uptimeLast30m": 99.644128113879, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | kwaipilot/kat-coder-pro-v2-20260327", "model_id": "kwaipilot/kat-coder-pro-v2", "model_name": "Kwaipilot: KAT-Coder-Pro V2", "context_length": 256000, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 80000, "max_prompt_tokens": null, "supported_parameters": [ "tool_choice", "tools", "max_tokens", "temperature", "top_p", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.644128113879, "uptime_last_5m": 100, "uptime_last_1d": 99.86144952462854, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "kwaipilot/kat-coder-pro-v2@atlascloud", "name": "Kwaipilot: KAT-Coder-Pro V2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "kwaipilot/kat-coder-pro-v2", "canonicalSlug": "kwaipilot/kat-coder-pro-v2-20260327", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 262144, "maxCompletionTokens": 144000, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | kwaipilot/kat-coder-pro-v2-20260327", "model_id": "kwaipilot/kat-coder-pro-v2", "model_name": "Kwaipilot: KAT-Coder-Pro V2", "context_length": 262144, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 144000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.89841201946213, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance/seedance-1-5-pro@seed", "name": "ByteDance: Seedance 1.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance/seedance-1-5-pro", "canonicalSlug": "bytedance/seedance-1-5-pro-20260320", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "frequency_penalty" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "ByteDance's next-generation audio-visual generation model with a 4.5B parameter Dual-Branch Diffusion Transformer architecture. Seedance 1.5 Pro generates video and audio simultaneously in a single unified pass — eliminating the timing...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance/seedance-1-5-pro-20260320", "model_id": "bytedance/seedance-1-5-pro", "model_name": "ByteDance: Seedance 1.5 Pro", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Seed", "tag": "seed", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "frequency_penalty" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/sora-2-pro@openai", "name": "OpenAI: Sora 2 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/sora-2-pro", "canonicalSlug": "openai/sora-2-pro-20260320", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "OpenAI's flagship video generation model, delivering production-quality video with physics-accurate motion, synchronized audio, and world-state persistence across shots. Sora 2 Pro follows intricate multi-shot instructions while maintaining consistent spatial relationships...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/sora-2-pro-20260320", "model_id": "openai/sora-2-pro", "model_name": "OpenAI: Sora 2 Pro", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/veo-3.1@google", "name": "Google: Veo 3.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "video", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/veo-3.1", "canonicalSlug": "google/veo-3.1-20260320", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text+image->video", "input_modalities": [ "text", "image" ], "output_modalities": [ "video" ], "tokenizer": "Other", "instruct_type": null }, "description": "Google's state-of-the-art video generation model, built for maximum visual fidelity in final production cuts. Veo 3.1 generates high-quality 1080p video from text or image prompts with native synchronized audio —...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/veo-3.1-20260320", "model_id": "google/veo-3.1", "model_name": "Google: Veo 3.1", "context_length": 0, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "rekaai/reka-edge@reka", "name": "Reka Edge", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "rekaai/reka-edge", "canonicalSlug": "rekaai/reka-edge-2603", "servingProvider": "Reka", "servingProviderSlug": "reka", "contextLength": 16384, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "tool_choice", "tools", "top_k", "top_p", "stop", "seed", "temperature", "frequency_penalty", "presence_penalty", "max_tokens", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Reka | rekaai/reka-edge-2603", "model_id": "rekaai/reka-edge", "model_name": "Reka Edge", "context_length": 16384, "pricing": { "prompt": "0.0000001", "completion": "0.0000001", "discount": 0 }, "provider_name": "Reka", "tag": "reka/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "tool_choice", "tools", "top_k", "top_p", "stop", "seed", "temperature", "frequency_penalty", "presence_penalty", "max_tokens", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.7@mara", "name": "MiniMax: MiniMax M2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000024" } ], "output": [ { "amount": 0.96, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000096" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.7", "canonicalSlug": "minimax/minimax-m2.7-20260318", "servingProvider": "Mara", "servingProviderSlug": "mara", "contextLength": 196608, "maxCompletionTokens": 196608, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "top_k", "top_p", "tool_choice", "tools", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mara | minimax/minimax-m2.7-20260318", "model_id": "minimax/minimax-m2.7", "model_name": "MiniMax: MiniMax M2.7", "context_length": 196608, "pricing": { "prompt": "0.00000024", "completion": "0.00000096", "discount": 0.6 }, "provider_name": "Mara", "tag": "mara", "quantization": "unknown", "max_completion_tokens": 196608, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "top_k", "top_p", "tool_choice", "tools", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9615827890895, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.7@gmicloud", "name": "MiniMax: MiniMax M2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000024" } ], "output": [ { "amount": 0.96, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000096" } ], "cacheRead": [ { "amount": 0.048, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000048" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.7", "canonicalSlug": "minimax/minimax-m2.7-20260318", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 196608, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | minimax/minimax-m2.7-20260318", "model_id": "minimax/minimax-m2.7", "model_name": "MiniMax: MiniMax M2.7", "context_length": 196608, "pricing": { "prompt": "0.00000024", "completion": "0.00000096", "input_cache_read": "0.000000048", "discount": 0.2 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.88877422313955, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.7@deepinfra", "name": "MiniMax: MiniMax M2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.7", "canonicalSlug": "minimax/minimax-m2.7-20260318", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 196608, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | minimax/minimax-m2.7-20260318", "model_id": "minimax/minimax-m2.7", "model_name": "MiniMax: MiniMax M2.7", "context_length": 196608, "pricing": { "prompt": "0.00000025", "completion": "0.000001", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.74822613870451, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.7@novita", "name": "MiniMax: MiniMax M2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 1.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000108" } ], "cacheRead": [ { "amount": 0.054, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000054" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.7", "canonicalSlug": "minimax/minimax-m2.7-20260318", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "uptimeLast30m": 99.28861788617887, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | minimax/minimax-m2.7-20260318", "model_id": "minimax/minimax-m2.7", "model_name": "MiniMax: MiniMax M2.7", "context_length": 204800, "pricing": { "prompt": "0.00000027", "completion": "0.00000108", "input_cache_read": "0.000000054", "discount": 0.1 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": 99.28861788617887, "uptime_last_5m": 100, "uptime_last_1d": 99.63499550763702, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.7@minimax", "name": "MiniMax: MiniMax M2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.7", "canonicalSlug": "minimax/minimax-m2.7-20260318", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tool_choice", "tools" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "uptimeLast30m": 97.53937007874016, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/minimax-m2.7-20260318", "model_id": "minimax/minimax-m2.7", "model_name": "MiniMax: MiniMax M2.7", "context_length": 204800, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 97.53937007874016, "uptime_last_5m": 100, "uptime_last_1d": 99.27732695551661, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.7@atlascloud", "name": "MiniMax: MiniMax M2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.7", "canonicalSlug": "minimax/minimax-m2.7-20260318", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 196608, "maxCompletionTokens": 196608, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "uptimeLast30m": 99.38650306748467, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | minimax/minimax-m2.7-20260318", "model_id": "minimax/minimax-m2.7", "model_name": "MiniMax: MiniMax M2.7", "context_length": 196608, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 196608, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.38650306748467, "uptime_last_5m": 100, "uptime_last_1d": 98.77536426208992, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.7@fireworks", "name": "MiniMax: MiniMax M2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.059, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000059" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.7", "canonicalSlug": "minimax/minimax-m2.7-20260318", "servingProvider": "Fireworks", "servingProviderSlug": "fireworks", "contextLength": 196608, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tool_choice", "tools" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fireworks | minimax/minimax-m2.7-20260318", "model_id": "minimax/minimax-m2.7", "model_name": "MiniMax: MiniMax M2.7", "context_length": 196608, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.000000059", "discount": 0 }, "provider_name": "Fireworks", "tag": "fireworks", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.7@deepinfra", "name": "MiniMax: MiniMax M2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000038" } ], "output": [ { "amount": 1.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000017" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.7", "canonicalSlug": "minimax/minimax-m2.7-20260318", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 196608, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | minimax/minimax-m2.7-20260318", "model_id": "minimax/minimax-m2.7", "model_name": "MiniMax: MiniMax M2.7", "context_length": 196608, "pricing": { "prompt": "0.00000038", "completion": "0.0000017", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/turbo", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 88.58653210788731, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.7@groq", "name": "MiniMax: MiniMax M2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.7", "canonicalSlug": "minimax/minimax-m2.7-20260318", "servingProvider": "Groq", "servingProviderSlug": "groq", "contextLength": 196608, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "uptimeLast30m": 99.0909090909091, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Groq | minimax/minimax-m2.7-20260318", "model_id": "minimax/minimax-m2.7", "model_name": "MiniMax: MiniMax M2.7", "context_length": 196608, "pricing": { "prompt": "0.0000006", "completion": "0.0000018", "discount": 0 }, "provider_name": "Groq", "tag": "groq", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.0909090909091, "uptime_last_5m": null, "uptime_last_1d": 98.66310160427807, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.7@minimax", "name": "MiniMax: MiniMax M2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.7", "canonicalSlug": "minimax/minimax-m2.7-20260318", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tool_choice", "tools" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/minimax-m2.7-20260318", "model_id": "minimax/minimax-m2.7", "model_name": "MiniMax: MiniMax M2.7", "context_length": 204800, "pricing": { "prompt": "0.0000006", "completion": "0.0000024", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax/highspeed", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 96.67777050740587, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.7@sambanova", "name": "MiniMax: MiniMax M2.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.7", "canonicalSlug": "minimax/minimax-m2.7-20260318", "servingProvider": "SambaNova", "servingProviderSlug": "sambanova", "contextLength": 196608, "maxCompletionTokens": 196608, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SambaNova | minimax/minimax-m2.7-20260318", "model_id": "minimax/minimax-m2.7", "model_name": "MiniMax: MiniMax M2.7", "context_length": 196608, "pricing": { "prompt": "0.0000006", "completion": "0.0000024", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "SambaNova", "tag": "sambanova", "quantization": "unknown", "max_completion_tokens": 196608, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.77343345416882, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-nano@azure", "name": "OpenAI: GPT-5.4 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-nano", "canonicalSlug": "openai/gpt-5.4-nano-20260317", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...", "uptimeLast30m": 95.01877346683354, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.4-nano-20260317", "model_id": "openai/gpt-5.4-nano", "model_name": "OpenAI: GPT-5.4 Nano", "context_length": 400000, "pricing": { "prompt": "0.0000002", "completion": "0.00000125", "web_search": "0.01", "input_cache_read": "0.00000002", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 95.01877346683354, "uptime_last_5m": 97.00729927007299, "uptime_last_1d": 99.92080354008986, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-nano@openai", "name": "OpenAI: GPT-5.4 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-nano", "canonicalSlug": "openai/gpt-5.4-nano-20260317", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...", "uptimeLast30m": 90.49199249747511, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-nano-20260317", "model_id": "openai/gpt-5.4-nano", "model_name": "OpenAI: GPT-5.4 Nano", "context_length": 400000, "pricing": { "prompt": "0.0000002", "completion": "0.00000125", "web_search": "0.01", "input_cache_read": "0.00000002", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": -2, "uptime_last_30m": 90.49199249747511, "uptime_last_5m": 90.0442140570209, "uptime_last_1d": 86.7139400213448, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-nano@openai", "name": "OpenAI: GPT-5.4 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-nano", "canonicalSlug": "openai/gpt-5.4-nano-20260317", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...", "uptimeLast30m": 90.49199249747511, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-nano-20260317", "model_id": "openai/gpt-5.4-nano", "model_name": "OpenAI: GPT-5.4 Nano", "context_length": 400000, "pricing": { "prompt": "0.0000001", "completion": "0.000000625", "web_search": "0.01", "input_cache_read": "0.00000001", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": -2, "uptime_last_30m": 90.49199249747511, "uptime_last_5m": 90.0442140570209, "uptime_last_1d": 86.7139400213448, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-nano@azure", "name": "OpenAI: GPT-5.4 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "cacheRead": [ { "amount": 0.022, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000022" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-nano", "canonicalSlug": "openai/gpt-5.4-nano-20260317", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...", "uptimeLast30m": 99.49087078651685, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.4-nano-20260317", "model_id": "openai/gpt-5.4-nano", "model_name": "OpenAI: GPT-5.4 Nano", "context_length": 400000, "pricing": { "prompt": "0.00000022", "completion": "0.000001375", "web_search": "0.01", "input_cache_read": "0.000000022", "discount": 0 }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.49087078651685, "uptime_last_5m": 100, "uptime_last_1d": 99.27785983915969, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-nano:batch@openai", "name": "OpenAI: GPT-5.4 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-nano:batch", "canonicalSlug": "openai/gpt-5.4-nano-20260317", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-nano-20260317:batch", "model_id": "openai/gpt-5.4-nano:batch", "model_name": "OpenAI: GPT-5.4 Nano", "context_length": 400000, "pricing": { "prompt": "0.0000001", "completion": "0.000000625", "web_search": "0.01", "input_cache_read": "0.00000001", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-nano:batch@openai", "name": "OpenAI: GPT-5.4 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.3125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003125" } ], "cacheRead": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-nano:batch", "canonicalSlug": "openai/gpt-5.4-nano-20260317", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-nano-20260317:batch", "model_id": "openai/gpt-5.4-nano:batch", "model_name": "OpenAI: GPT-5.4 Nano", "context_length": 400000, "pricing": { "prompt": "0.00000005", "completion": "0.0000003125", "web_search": "0.01", "input_cache_read": "0.000000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-mini@openai", "name": "OpenAI: GPT-5.4 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-mini", "canonicalSlug": "openai/gpt-5.4-mini-20260317", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...", "uptimeLast30m": 99.8971732339988, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-mini-20260317", "model_id": "openai/gpt-5.4-mini", "model_name": "OpenAI: GPT-5.4 Mini", "context_length": 400000, "pricing": { "prompt": "0.00000075", "completion": "0.0000045", "web_search": "0.01", "input_cache_read": "0.000000075", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.8971732339988, "uptime_last_5m": 100, "uptime_last_1d": 99.318898516692, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-mini@openai", "name": "OpenAI: GPT-5.4 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "output": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-mini", "canonicalSlug": "openai/gpt-5.4-mini-20260317", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...", "uptimeLast30m": 99.8971732339988, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-mini-20260317", "model_id": "openai/gpt-5.4-mini", "model_name": "OpenAI: GPT-5.4 Mini", "context_length": 400000, "pricing": { "prompt": "0.000000375", "completion": "0.00000225", "web_search": "0.01", "input_cache_read": "0.0000000375", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.8971732339988, "uptime_last_5m": 100, "uptime_last_1d": 99.318898516692, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-mini@openai", "name": "OpenAI: GPT-5.4 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-mini", "canonicalSlug": "openai/gpt-5.4-mini-20260317", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...", "uptimeLast30m": 99.8971732339988, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-mini-20260317", "model_id": "openai/gpt-5.4-mini", "model_name": "OpenAI: GPT-5.4 Mini", "context_length": 400000, "pricing": { "prompt": "0.0000015", "completion": "0.000009", "web_search": "0.01", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.8971732339988, "uptime_last_5m": 100, "uptime_last_1d": 99.318898516692, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-mini@azure", "name": "OpenAI: GPT-5.4 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-mini", "canonicalSlug": "openai/gpt-5.4-mini-20260317", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.4-mini-20260317", "model_id": "openai/gpt-5.4-mini", "model_name": "OpenAI: GPT-5.4 Mini", "context_length": 400000, "pricing": { "prompt": "0.00000075", "completion": "0.0000045", "web_search": "0.01", "input_cache_read": "0.000000075", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9889079917919, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-mini@azure", "name": "OpenAI: GPT-5.4 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.8250000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000825" } ], "output": [ { "amount": 4.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000495" } ], "cacheRead": [ { "amount": 0.0825, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000825" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-mini", "canonicalSlug": "openai/gpt-5.4-mini-20260317", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.4-mini-20260317", "model_id": "openai/gpt-5.4-mini", "model_name": "OpenAI: GPT-5.4 Mini", "context_length": 400000, "pricing": { "prompt": "0.000000825", "completion": "0.00000495", "web_search": "0.01", "input_cache_read": "0.0000000825", "discount": 0 }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.74396275821937, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-mini:batch@openai", "name": "OpenAI: GPT-5.4 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "output": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-mini:batch", "canonicalSlug": "openai/gpt-5.4-mini-20260317", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-mini-20260317:batch", "model_id": "openai/gpt-5.4-mini:batch", "model_name": "OpenAI: GPT-5.4 Mini", "context_length": 400000, "pricing": { "prompt": "0.000000375", "completion": "0.00000225", "web_search": "0.01", "input_cache_read": "0.0000000375", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-mini:batch@openai", "name": "OpenAI: GPT-5.4 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "output": [ { "amount": 1.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001125" } ], "cacheRead": [ { "amount": 0.01875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001875" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-mini:batch", "canonicalSlug": "openai/gpt-5.4-mini-20260317", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-mini-20260317:batch", "model_id": "openai/gpt-5.4-mini:batch", "model_name": "OpenAI: GPT-5.4 Mini", "context_length": 400000, "pricing": { "prompt": "0.0000001875", "completion": "0.000001125", "web_search": "0.01", "input_cache_read": "0.00000001875", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-mini:batch@openai", "name": "OpenAI: GPT-5.4 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-mini:batch", "canonicalSlug": "openai/gpt-5.4-mini-20260317", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-mini-20260317:batch", "model_id": "openai/gpt-5.4-mini:batch", "model_name": "OpenAI: GPT-5.4 Mini", "context_length": 400000, "pricing": { "prompt": "0.00000075", "completion": "0.0000045", "web_search": "0.01", "input_cache_read": "0.000000075", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-small-2603@mistral", "name": "Mistral: Mistral Small 4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-small-2603", "canonicalSlug": "mistralai/mistral-small-2603", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...", "uptimeLast30m": 99.94813278008299, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/mistral-small-2603", "model_id": "mistralai/mistral-small-2603", "model_name": "Mistral: Mistral Small 4", "context_length": 262144, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.000000015", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.94813278008299, "uptime_last_5m": 100, "uptime_last_1d": 99.95755793047753, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-small-2603@venice", "name": "Mistral: Mistral Small 4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-small-2603", "canonicalSlug": "mistralai/mistral-small-2603", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...", "uptimeLast30m": 99.66703662597114, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | mistralai/mistral-small-2603", "model_id": "mistralai/mistral-small-2603", "model_name": "Mistral: Mistral Small 4", "context_length": 256000, "pricing": { "prompt": "0.0000001875", "completion": "0.00000075", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.66703662597114, "uptime_last_5m": 99.28057553956835, "uptime_last_1d": 99.87848722863191, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "perplexity/pplx-embed-v1-4b@perplexity", "name": "Perplexity: Embed V1 4B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000003" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "perplexity/pplx-embed-v1-4b", "canonicalSlug": "perplexity/pplx-embed-v1-4B", "servingProvider": "Perplexity", "servingProviderSlug": "perplexity", "contextLength": 32000, "quantization": "int8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "pplx-embed-v1 -4B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 4B parameter model maximizing retrieval...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Perplexity | perplexity/pplx-embed-v1-4B", "model_id": "perplexity/pplx-embed-v1-4b", "model_name": "Perplexity: Embed V1 4B", "context_length": 32000, "pricing": { "prompt": "0.00000003", "completion": "0", "discount": 0 }, "provider_name": "Perplexity", "tag": "perplexity/int8", "quantization": "int8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "perplexity/pplx-embed-v1-0.6b@perplexity", "name": "Perplexity: Embed V1 0.6B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.000000004" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "perplexity/pplx-embed-v1-0.6b", "canonicalSlug": "perplexity/pplx-embed-v1-0.6B", "servingProvider": "Perplexity", "servingProviderSlug": "perplexity", "contextLength": 32000, "quantization": "int8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "pplx-embed-v1-0.6B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 0.6B parameter model targeting lightweight, low-latency...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Perplexity | perplexity/pplx-embed-v1-0.6B", "model_id": "perplexity/pplx-embed-v1-0.6b", "model_name": "Perplexity: Embed V1 0.6B", "context_length": 32000, "pricing": { "prompt": "0.000000004", "completion": "0", "discount": 0 }, "provider_name": "Perplexity", "tag": "perplexity/int8", "quantization": "int8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5-turbo@z.ai", "name": "Z.ai: GLM 5 Turbo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000024" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5-turbo", "canonicalSlug": "z-ai/glm-5-turbo-20260315", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "top_k" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-5-turbo-20260315", "model_id": "z-ai/glm-5-turbo", "model_name": "Z.ai: GLM 5 Turbo", "context_length": 202752, "pricing": { "prompt": "0.0000012", "completion": "0.000004", "input_cache_read": "0.00000024", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "top_k" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-super-120b-a12b@deepinfra", "name": "NVIDIA: Nemotron 3 Super", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000085" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-super-120b-a12b", "canonicalSlug": "nvidia/nemotron-3-super-120b-a12b-20230311", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "bf16", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...", "uptimeLast30m": 88.57938718662952, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | nvidia/nemotron-3-super-120b-a12b-20230311", "model_id": "nvidia/nemotron-3-super-120b-a12b", "model_name": "NVIDIA: Nemotron 3 Super", "context_length": 262144, "pricing": { "prompt": "0.000000085", "completion": "0.0000004", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "reasoning_effort" ], "status": -2, "uptime_last_30m": 88.57938718662952, "uptime_last_5m": 90, "uptime_last_1d": 94.47887061454597, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-super-120b-a12b@digitalocean", "name": "NVIDIA: Nemotron 3 Super", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.165, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000165" } ], "output": [ { "amount": 0.35750000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003575" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-super-120b-a12b", "canonicalSlug": "nvidia/nemotron-3-super-120b-a12b-20230311", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 1000000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "structured_outputs", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | nvidia/nemotron-3-super-120b-a12b-20230311", "model_id": "nvidia/nemotron-3-super-120b-a12b", "model_name": "NVIDIA: Nemotron 3 Super", "context_length": 1000000, "pricing": { "prompt": "0.000000165", "completion": "0.0000003575", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "structured_outputs", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.39448897209384, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-super-120b-a12b@nebius", "name": "NVIDIA: Nemotron 3 Super", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-super-120b-a12b", "canonicalSlug": "nvidia/nemotron-3-super-120b-a12b-20230311", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 8000, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "response_format", "repetition_penalty", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | nvidia/nemotron-3-super-120b-a12b-20230311", "model_id": "nvidia/nemotron-3-super-120b-a12b", "model_name": "NVIDIA: Nemotron 3 Super", "context_length": 8000, "pricing": { "prompt": "0.0000003", "completion": "0.0000009", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/fp4", "quantization": "fp4", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "response_format", "repetition_penalty", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.94405309869542, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-super-120b-a12b:free@nvidia", "name": "NVIDIA: Nemotron 3 Super", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-super-120b-a12b:free", "canonicalSlug": "nvidia/nemotron-3-super-120b-a12b-20230311", "servingProvider": "Nvidia", "servingProviderSlug": "nvidia", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tools", "tool_choice", "structured_outputs", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...", "uptimeLast30m": 99.87820968233025, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nvidia | nvidia/nemotron-3-super-120b-a12b-20230311:free", "model_id": "nvidia/nemotron-3-super-120b-a12b:free", "model_name": "NVIDIA: Nemotron 3 Super", "context_length": 262144, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Nvidia", "tag": "nvidia", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tools", "tool_choice", "structured_outputs", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.87820968233025, "uptime_last_5m": 99.91686065846358, "uptime_last_1d": 99.74983379284723, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance-seed/seed-2.0-lite@seed", "name": "ByteDance Seed: Seed-2.0-Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance-seed/seed-2.0-lite", "canonicalSlug": "bytedance-seed/seed-2.0-lite-20260309", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "structured_outputs", "tool_choice", "tools", "response_format", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance-seed/seed-2.0-lite-20260309", "model_id": "bytedance-seed/seed-2.0-lite", "model_name": "ByteDance Seed: Seed-2.0-Lite", "context_length": 262144, "pricing": { "prompt": "0.00000025", "completion": "0.000002", "discount": 0, "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.0000005", "completion": "0.000004" } ] }, "provider_name": "Seed", "tag": "seed/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "structured_outputs", "tool_choice", "tools", "response_format", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.99200788555292, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-9b@siliconflow", "name": "Qwen: Qwen3.5-9B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-9b", "canonicalSlug": "qwen/qwen3.5-9b-20260310", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...", "uptimeLast30m": 96.87012017109105, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3.5-9b-20260310", "model_id": "qwen/qwen3.5-9b", "model_name": "Qwen: Qwen3.5-9B", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.00000015", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 96.87012017109105, "uptime_last_5m": 98.93795416433761, "uptime_last_1d": 98.87910749592163, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-9b@deepinfra", "name": "Qwen: Qwen3.5-9B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-9b", "canonicalSlug": "qwen/qwen3.5-9b-20260310", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 81920, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...", "uptimeLast30m": 98.99919935948759, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3.5-9b-20260310", "model_id": "qwen/qwen3.5-9b", "model_name": "Qwen: Qwen3.5-9B", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.00000015", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 81920, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 98.99919935948759, "uptime_last_5m": 99.06976744186046, "uptime_last_1d": 97.4478129925484, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-9b@venice", "name": "Qwen: Qwen3.5-9B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-9b", "canonicalSlug": "qwen/qwen3.5-9b-20260310", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | qwen/qwen3.5-9b-20260310", "model_id": "qwen/qwen3.5-9b", "model_name": "Qwen: Qwen3.5-9B", "context_length": 256000, "pricing": { "prompt": "0.0000001", "completion": "0.00000015", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.92576999778417, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-9b@parasail", "name": "Qwen: Qwen3.5-9B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-9b", "canonicalSlug": "qwen/qwen3.5-9b-20260310", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | qwen/qwen3.5-9b-20260310", "model_id": "qwen/qwen3.5-9b", "model_name": "Qwen: Qwen3.5-9B", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.00000025", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.61382506275342, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-9b@together", "name": "Qwen: Qwen3.5-9B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000017" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-9b", "canonicalSlug": "qwen/qwen3.5-9b-20260310", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | qwen/qwen3.5-9b-20260310", "model_id": "qwen/qwen3.5-9b", "model_name": "Qwen: Qwen3.5-9B", "context_length": 262144, "pricing": { "prompt": "0.00000017", "completion": "0.00000025", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.79357661499698, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-pro@openai", "name": "OpenAI: GPT-5.4 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-pro", "canonicalSlug": "openai/gpt-5.4-pro-20260305", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-pro-20260305", "model_id": "openai/gpt-5.4-pro", "model_name": "OpenAI: GPT-5.4 Pro", "context_length": 1050000, "pricing": { "prompt": "0.00003", "completion": "0.00018", "web_search": "0.01", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00006", "completion": "0.00027" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.66457023060796, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-pro@openai", "name": "OpenAI: GPT-5.4 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-pro", "canonicalSlug": "openai/gpt-5.4-pro-20260305", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-pro-20260305", "model_id": "openai/gpt-5.4-pro", "model_name": "OpenAI: GPT-5.4 Pro", "context_length": 1050000, "pricing": { "prompt": "0.000015", "completion": "0.00009", "web_search": "0.01", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00003", "completion": "0.000135" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.66457023060796, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-pro@azure", "name": "OpenAI: GPT-5.4 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-pro", "canonicalSlug": "openai/gpt-5.4-pro-20260305", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.4-pro-20260305", "model_id": "openai/gpt-5.4-pro", "model_name": "OpenAI: GPT-5.4 Pro", "context_length": 1050000, "pricing": { "prompt": "0.00003", "completion": "0.00018", "web_search": "0.01", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00006", "completion": "0.00027" } ] }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-pro:batch@openai", "name": "OpenAI: GPT-5.4 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-pro:batch", "canonicalSlug": "openai/gpt-5.4-pro-20260305", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-pro-20260305:batch", "model_id": "openai/gpt-5.4-pro:batch", "model_name": "OpenAI: GPT-5.4 Pro", "context_length": 1050000, "pricing": { "prompt": "0.000015", "completion": "0.00009", "web_search": "0.01", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00003", "completion": "0.000135" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4-pro:batch@openai", "name": "OpenAI: GPT-5.4 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "output": [ { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4-pro:batch", "canonicalSlug": "openai/gpt-5.4-pro-20260305", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-pro-20260305:batch", "model_id": "openai/gpt-5.4-pro:batch", "model_name": "OpenAI: GPT-5.4 Pro", "context_length": 1050000, "pricing": { "prompt": "0.0000075", "completion": "0.000045", "web_search": "0.01", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000015", "completion": "0.0000675" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4@openai", "name": "OpenAI: GPT-5.4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4", "canonicalSlug": "openai/gpt-5.4-20260305", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": 99.63327674023769, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-20260305", "model_id": "openai/gpt-5.4", "model_name": "OpenAI: GPT-5.4", "context_length": 1050000, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000005", "completion": "0.0000225", "input_cache_read": "0.0000005" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.63327674023769, "uptime_last_5m": 99.54721862871928, "uptime_last_1d": 99.70474750727216, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4@openai", "name": "OpenAI: GPT-5.4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4", "canonicalSlug": "openai/gpt-5.4-20260305", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": 99.63327674023769, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-20260305", "model_id": "openai/gpt-5.4", "model_name": "OpenAI: GPT-5.4", "context_length": 1050000, "pricing": { "prompt": "0.00000125", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.000000125", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000025", "completion": "0.00001125", "input_cache_read": "0.00000025" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.63327674023769, "uptime_last_5m": 99.54721862871928, "uptime_last_1d": 99.70474750727216, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4@openai", "name": "OpenAI: GPT-5.4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4", "canonicalSlug": "openai/gpt-5.4-20260305", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": 99.63327674023769, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-20260305", "model_id": "openai/gpt-5.4", "model_name": "OpenAI: GPT-5.4", "context_length": 1050000, "pricing": { "prompt": "0.000005", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.0000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.63327674023769, "uptime_last_5m": 99.54721862871928, "uptime_last_1d": 99.70474750727216, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4@azure", "name": "OpenAI: GPT-5.4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4", "canonicalSlug": "openai/gpt-5.4-20260305", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": 99.91666666666667, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.4-20260305", "model_id": "openai/gpt-5.4", "model_name": "OpenAI: GPT-5.4", "context_length": 1050000, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.000005", "completion": "0.0000225", "input_cache_read": "0.0000005" } ] }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.91666666666667, "uptime_last_5m": 100, "uptime_last_1d": 99.97360519977563, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4@azure", "name": "OpenAI: GPT-5.4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4", "canonicalSlug": "openai/gpt-5.4-20260305", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.4-20260305", "model_id": "openai/gpt-5.4", "model_name": "OpenAI: GPT-5.4", "context_length": 1050000, "pricing": { "prompt": "0.00000275", "completion": "0.0000165", "web_search": "0.01", "input_cache_read": "0.000000275", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000055", "completion": "0.00002475", "input_cache_read": "0.00000055" } ] }, "provider_name": "Azure", "tag": "azure/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4@azure", "name": "OpenAI: GPT-5.4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4", "canonicalSlug": "openai/gpt-5.4-20260305", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.4-20260305", "model_id": "openai/gpt-5.4", "model_name": "OpenAI: GPT-5.4", "context_length": 1050000, "pricing": { "prompt": "0.00000275", "completion": "0.0000165", "web_search": "0.01", "input_cache_read": "0.000000275", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000055", "completion": "0.00002475", "input_cache_read": "0.00000055" } ] }, "provider_name": "Azure", "tag": "azure/eu", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4@amazon-bedrock", "name": "OpenAI: GPT-5.4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4", "canonicalSlug": "openai/gpt-5.4-20260305", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | openai/gpt-5.4-20260305", "model_id": "openai/gpt-5.4", "model_name": "OpenAI: GPT-5.4", "context_length": 1050000, "pricing": { "prompt": "0.00000275", "completion": "0.0000165", "web_search": "0.01", "input_cache_read": "0.000000275", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000055", "completion": "0.00002475", "input_cache_read": "0.00000055" } ] }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/us-east-1", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 922000, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4:batch@openai", "name": "OpenAI: GPT-5.4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4:batch", "canonicalSlug": "openai/gpt-5.4-20260305", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-20260305:batch", "model_id": "openai/gpt-5.4:batch", "model_name": "OpenAI: GPT-5.4", "context_length": 1050000, "pricing": { "prompt": "0.00000125", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.000000125", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.0000025", "completion": "0.00001125", "input_cache_read": "0.00000025" } ] }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4:batch@openai", "name": "OpenAI: GPT-5.4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4:batch", "canonicalSlug": "openai/gpt-5.4-20260305", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-20260305:batch", "model_id": "openai/gpt-5.4:batch", "model_name": "OpenAI: GPT-5.4", "context_length": 1050000, "pricing": { "prompt": "0.000000625", "completion": "0.00000375", "web_search": "0.01", "input_cache_read": "0.0000000625", "discount": 0, "overrides": [ { "min_prompt_tokens": 272000, "prompt": "0.00000125", "completion": "0.000005625", "input_cache_read": "0.000000125" } ] }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.4:batch@openai", "name": "OpenAI: GPT-5.4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.4:batch", "canonicalSlug": "openai/gpt-5.4-20260305", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1050000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.4-20260305:batch", "model_id": "openai/gpt-5.4:batch", "model_name": "OpenAI: GPT-5.4", "context_length": 1050000, "pricing": { "prompt": "0.0000025", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "inception/mercury-2@inception", "name": "Inception: Mercury 2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "inception/mercury-2", "canonicalSlug": "inception/mercury-2-20260304", "servingProvider": "Inception", "servingProviderSlug": "inception", "contextLength": 128000, "maxCompletionTokens": 50000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "temperature", "tools", "tool_choice", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Inception | inception/mercury-2-20260304", "model_id": "inception/mercury-2", "model_name": "Inception: Mercury 2", "context_length": 128000, "pricing": { "prompt": "0.00000025", "completion": "0.00000075", "input_cache_read": "0.000000025", "discount": 0 }, "provider_name": "Inception", "tag": "inception", "quantization": "unknown", "max_completion_tokens": 50000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "temperature", "tools", "tool_choice", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.98478058679294, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite-preview@google-ai-studio", "name": "Google: Gemini 3.1 Flash Lite Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 2.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000025" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite-preview", "canonicalSlug": "google/gemini-3.1-flash-lite-preview-20260303", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...", "uptimeLast30m": 98.51900393184798, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-flash-lite-preview-20260303", "model_id": "google/gemini-3.1-flash-lite-preview", "model_name": "Google: Gemini 3.1 Flash Lite Preview", "context_length": 1048576, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image": "0.00000025", "audio": "0.0000005", "input_audio_cache": "0.00000005", "web_search": "0.014", "internal_reasoning": "0.0000015", "input_cache_read": "0.000000025", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.51900393184798, "uptime_last_5m": 98.27062397756485, "uptime_last_1d": 98.89327033301389, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite-preview@google-ai-studio", "name": "Google: Gemini 3.1 Flash Lite Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [ { "amount": 0.012499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000125" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 1.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite-preview", "canonicalSlug": "google/gemini-3.1-flash-lite-preview-20260303", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...", "uptimeLast30m": 98.51900393184798, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-flash-lite-preview-20260303", "model_id": "google/gemini-3.1-flash-lite-preview", "model_name": "Google: Gemini 3.1 Flash Lite Preview", "context_length": 1048576, "pricing": { "prompt": "0.000000125", "completion": "0.00000075", "image": "0.000000125", "audio": "0.00000025", "input_audio_cache": "0.000000025", "web_search": "0.014", "internal_reasoning": "0.00000075", "input_cache_read": "0.0000000125", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.51900393184798, "uptime_last_5m": 98.27062397756485, "uptime_last_1d": 98.89327033301389, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-lite-preview@google-ai-studio", "name": "Google: Gemini 3.1 Flash Lite Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "cacheRead": [ { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000045" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 4.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000045" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-lite-preview", "canonicalSlug": "google/gemini-3.1-flash-lite-preview-20260303", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "video", "file", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...", "uptimeLast30m": 98.51900393184798, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-flash-lite-preview-20260303", "model_id": "google/gemini-3.1-flash-lite-preview", "model_name": "Google: Gemini 3.1 Flash Lite Preview", "context_length": 1048576, "pricing": { "prompt": "0.00000045", "completion": "0.0000027", "image": "0.00000045", "audio": "0.0000009", "input_audio_cache": "0.00000009", "web_search": "0.014", "internal_reasoning": "0.0000027", "input_cache_read": "0.000000045", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.51900393184798, "uptime_last_5m": 98.27062397756485, "uptime_last_1d": 98.89327033301389, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance-seed/seed-2.0-mini@seed", "name": "ByteDance Seed: Seed-2.0-Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance-seed/seed-2.0-mini", "canonicalSlug": "bytedance-seed/seed-2.0-mini-20260224", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "structured_outputs", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance-seed/seed-2.0-mini-20260224", "model_id": "bytedance-seed/seed-2.0-mini", "model_name": "ByteDance Seed: Seed-2.0-Mini", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.0000004", "discount": 0, "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.0000002", "completion": "0.0000008" } ] }, "provider_name": "Seed", "tag": "seed/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "structured_outputs", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-flash-image-preview@google-ai-studio", "name": "Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000005" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-flash-image-preview", "canonicalSlug": "google/gemini-3.1-flash-image-preview-20260226", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 65536, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-flash-image-preview-20260226", "model_id": "google/gemini-3.1-flash-image-preview", "model_name": "Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)", "context_length": 65536, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "image_output": "0.00006", "web_search": "0.014", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99567048025197, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-35b-a3b@deepinfra", "name": "Qwen: Qwen3.5-35B-A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-35b-a3b", "canonicalSlug": "qwen/qwen3.5-35b-a3b-20260224", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 81920, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3.5-35b-a3b-20260224", "model_id": "qwen/qwen3.5-35b-a3b", "model_name": "Qwen: Qwen3.5-35B-A3B", "context_length": 262144, "pricing": { "prompt": "0.00000014", "completion": "0.000001", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 81920, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.70465041348943, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-35b-a3b@akashml", "name": "Qwen: Qwen3.5-35B-A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-35b-a3b", "canonicalSlug": "qwen/qwen3.5-35b-a3b-20260224", "servingProvider": "AkashML", "servingProviderSlug": "akashml", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "max_tokens", "structured_outputs", "tools", "response_format", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...", "uptimeLast30m": 99.70501474926253, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AkashML | qwen/qwen3.5-35b-a3b-20260224", "model_id": "qwen/qwen3.5-35b-a3b", "model_name": "Qwen: Qwen3.5-35B-A3B", "context_length": 262144, "pricing": { "prompt": "0.00000014", "completion": "0.000001", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "AkashML", "tag": "akashml/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "max_tokens", "structured_outputs", "tools", "response_format", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.70501474926253, "uptime_last_5m": null, "uptime_last_1d": 98.837591121035, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-35b-a3b@parasail", "name": "Qwen: Qwen3.5-35B-A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-35b-a3b", "canonicalSlug": "qwen/qwen3.5-35b-a3b-20260224", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | qwen/qwen3.5-35b-a3b-20260224", "model_id": "qwen/qwen3.5-35b-a3b", "model_name": "Qwen: Qwen3.5-35B-A3B", "context_length": 262144, "pricing": { "prompt": "0.00000015", "completion": "0.000001", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.72323127486594, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-35b-a3b@alibaba", "name": "Qwen: Qwen3.5-35B-A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.1625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001625" } ], "output": [ { "amount": 1.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000013" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-35b-a3b", "canonicalSlug": "qwen/qwen3.5-35b-a3b-20260224", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "top_logprobs", "logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...", "uptimeLast30m": 99.49238578680203, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.5-35b-a3b-20260224", "model_id": "qwen/qwen3.5-35b-a3b", "model_name": "Qwen: Qwen3.5-35B-A3B", "context_length": 262144, "pricing": { "prompt": "0.0000001625", "completion": "0.0000013", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 258048, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "top_logprobs", "logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 99.49238578680203, "uptime_last_5m": null, "uptime_last_1d": 99.54749103942653, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-35b-a3b@atlascloud", "name": "Qwen: Qwen3.5-35B-A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000225" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [ { "amount": 0.22499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000225" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-35b-a3b", "canonicalSlug": "qwen/qwen3.5-35b-a3b-20260224", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...", "uptimeLast30m": 99.85207100591717, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | qwen/qwen3.5-35b-a3b-20260224", "model_id": "qwen/qwen3.5-35b-a3b", "model_name": "Qwen: Qwen3.5-35B-A3B", "context_length": 262144, "pricing": { "prompt": "0.000000225", "completion": "0.0000018", "input_cache_read": "0.000000225", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.85207100591717, "uptime_last_5m": null, "uptime_last_1d": 99.91327453223604, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-35b-a3b@siliconflow", "name": "Qwen: Qwen3.5-35B-A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000024" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-35b-a3b", "canonicalSlug": "qwen/qwen3.5-35b-a3b-20260224", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3.5-35b-a3b-20260224", "model_id": "qwen/qwen3.5-35b-a3b", "model_name": "Qwen: Qwen3.5-35B-A3B", "context_length": 262144, "pricing": { "prompt": "0.00000024", "completion": "0.0000018", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 95.74322004806042, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-35b-a3b@coreweave", "name": "Qwen: Qwen3.5-35B-A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-35b-a3b", "canonicalSlug": "qwen/qwen3.5-35b-a3b-20260224", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | qwen/qwen3.5-35b-a3b-20260224", "model_id": "qwen/qwen3.5-35b-a3b", "model_name": "Qwen: Qwen3.5-35B-A3B", "context_length": 262144, "pricing": { "prompt": "0.00000025", "completion": "0.00000125", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.87985351370709, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-35b-a3b@venice", "name": "Qwen: Qwen3.5-35B-A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003125" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.15625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015625" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-35b-a3b", "canonicalSlug": "qwen/qwen3.5-35b-a3b-20260224", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tool_choice", "logprobs", "response_format", "structured_outputs", "tools", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | qwen/qwen3.5-35b-a3b-20260224", "model_id": "qwen/qwen3.5-35b-a3b", "model_name": "Qwen: Qwen3.5-35B-A3B", "context_length": 256000, "pricing": { "prompt": "0.0000003125", "completion": "0.00000125", "input_cache_read": "0.00000015625", "discount": 0 }, "provider_name": "Venice", "tag": "venice", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tool_choice", "logprobs", "response_format", "structured_outputs", "tools", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96245053580198, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-27b@alibaba", "name": "Qwen: Qwen3.5-27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.195, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000195" } ], "output": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-27b", "canonicalSlug": "qwen/qwen3.5-27b-20260224", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "top_logprobs", "logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...", "uptimeLast30m": 99.67234600262124, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.5-27b-20260224", "model_id": "qwen/qwen3.5-27b", "model_name": "Qwen: Qwen3.5-27B", "context_length": 262144, "pricing": { "prompt": "0.000000195", "completion": "0.00000156", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 258048, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "top_logprobs", "logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 99.67234600262124, "uptime_last_5m": 100, "uptime_last_1d": 96.84242485578383, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-27b@siliconflow", "name": "Qwen: Qwen3.5-27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-27b", "canonicalSlug": "qwen/qwen3.5-27b-20260224", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...", "uptimeLast30m": 99.34282584884994, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3.5-27b-20260224", "model_id": "qwen/qwen3.5-27b", "model_name": "Qwen: Qwen3.5-27B", "context_length": 262144, "pricing": { "prompt": "0.00000025", "completion": "0.000002", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.34282584884994, "uptime_last_5m": 100, "uptime_last_1d": 98.13881079071497, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-27b@deepinfra", "name": "Qwen: Qwen3.5-27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "output": [ { "amount": 2.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000026" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-27b", "canonicalSlug": "qwen/qwen3.5-27b-20260224", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 81920, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...", "uptimeLast30m": 98.00995024875621, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3.5-27b-20260224", "model_id": "qwen/qwen3.5-27b", "model_name": "Qwen: Qwen3.5-27B", "context_length": 262144, "pricing": { "prompt": "0.00000026", "completion": "0.0000026", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 81920, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 98.00995024875621, "uptime_last_5m": 96.84210526315789, "uptime_last_1d": 96.21384952183432, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-27b@atlascloud", "name": "Qwen: Qwen3.5-27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 2.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000216" } ], "cacheRead": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-27b", "canonicalSlug": "qwen/qwen3.5-27b-20260224", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...", "uptimeLast30m": 96.91275167785236, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | qwen/qwen3.5-27b-20260224", "model_id": "qwen/qwen3.5-27b", "model_name": "Qwen: Qwen3.5-27B", "context_length": 262144, "pricing": { "prompt": "0.00000027", "completion": "0.00000216", "input_cache_read": "0.00000027", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 96.91275167785236, "uptime_last_5m": 98.9247311827957, "uptime_last_1d": 97.4114657773543, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-27b@novita", "name": "Qwen: Qwen3.5-27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-27b", "canonicalSlug": "qwen/qwen3.5-27b-20260224", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...", "uptimeLast30m": 99.83443708609272, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3.5-27b-20260224", "model_id": "qwen/qwen3.5-27b", "model_name": "Qwen: Qwen3.5-27B", "context_length": 262144, "pricing": { "prompt": "0.0000003", "completion": "0.0000024", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.83443708609272, "uptime_last_5m": 100, "uptime_last_1d": 99.84056987788331, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-27b@phala", "name": "Qwen: Qwen3.5-27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-27b", "canonicalSlug": "qwen/qwen3.5-27b-20260224", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "logprobs", "top_logprobs", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | qwen/qwen3.5-27b-20260224", "model_id": "qwen/qwen3.5-27b", "model_name": "Qwen: Qwen3.5-27B", "context_length": 262144, "pricing": { "prompt": "0.0000003", "completion": "0.0000024", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "logprobs", "top_logprobs", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 97.13304044159062, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-122b-a10b@siliconflow", "name": "Qwen: Qwen3.5-122B-A10B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "output": [ { "amount": 2.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000208" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-122b-a10b", "canonicalSlug": "qwen/qwen3.5-122b-a10b-20260224", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...", "uptimeLast30m": 99.3920972644377, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3.5-122b-a10b-20260224", "model_id": "qwen/qwen3.5-122b-a10b", "model_name": "Qwen: Qwen3.5-122B-A10B", "context_length": 262144, "pricing": { "prompt": "0.00000026", "completion": "0.00000208", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "repetition_penalty", "presence_penalty", "max_tokens", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.3920972644377, "uptime_last_5m": null, "uptime_last_1d": 98.7173176185989, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-122b-a10b@alibaba", "name": "Qwen: Qwen3.5-122B-A10B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "output": [ { "amount": 2.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000208" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-122b-a10b", "canonicalSlug": "qwen/qwen3.5-122b-a10b-20260224", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...", "uptimeLast30m": 99.78118161925602, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.5-122b-a10b-20260224", "model_id": "qwen/qwen3.5-122b-a10b", "model_name": "Qwen: Qwen3.5-122B-A10B", "context_length": 262144, "pricing": { "prompt": "0.00000026", "completion": "0.00000208", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 258048, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 99.78118161925602, "uptime_last_5m": null, "uptime_last_1d": 99.76677242451622, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-122b-a10b@deepinfra", "name": "Qwen: Qwen3.5-122B-A10B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000029" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-122b-a10b", "canonicalSlug": "qwen/qwen3.5-122b-a10b-20260224", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 81920, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3.5-122b-a10b-20260224", "model_id": "qwen/qwen3.5-122b-a10b", "model_name": "Qwen: Qwen3.5-122B-A10B", "context_length": 262144, "pricing": { "prompt": "0.00000029", "completion": "0.0000024", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 81920, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.82828021714244, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-122b-a10b@atlascloud", "name": "Qwen: Qwen3.5-122B-A10B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-122b-a10b", "canonicalSlug": "qwen/qwen3.5-122b-a10b-20260224", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | qwen/qwen3.5-122b-a10b-20260224", "model_id": "qwen/qwen3.5-122b-a10b", "model_name": "Qwen: Qwen3.5-122B-A10B", "context_length": 262144, "pricing": { "prompt": "0.0000003", "completion": "0.0000024", "input_cache_read": "0.0000003", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 98.961672981638, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-122b-a10b@novita", "name": "Qwen: Qwen3.5-122B-A10B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-122b-a10b", "canonicalSlug": "qwen/qwen3.5-122b-a10b-20260224", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3.5-122b-a10b-20260224", "model_id": "qwen/qwen3.5-122b-a10b", "model_name": "Qwen: Qwen3.5-122B-A10B", "context_length": 262144, "pricing": { "prompt": "0.0000004", "completion": "0.0000032", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.08262399615003, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-flash-02-23@alibaba", "name": "Qwen: Qwen3.5-Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.065, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000065" } ], "output": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-flash-02-23", "canonicalSlug": "qwen/qwen3.5-flash-20260224", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.5-flash-20260224", "model_id": "qwen/qwen3.5-flash-02-23", "model_name": "Qwen: Qwen3.5-Flash", "context_length": 1000000, "pricing": { "prompt": "0.000000065", "completion": "0.00000026", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 983616, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99988480520128, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-pro-preview-customtools@google-ai-studio", "name": "Google: Gemini 3.1 Pro Preview Custom Tools", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.000002, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000002" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-pro-preview-customtools", "canonicalSlug": "google/gemini-3.1-pro-preview-customtools-20260219", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "audio", "image", "video", "file" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-pro-preview-customtools-20260219", "model_id": "google/gemini-3.1-pro-preview-customtools", "model_name": "Google: Gemini 3.1 Pro Preview Custom Tools", "context_length": 1048576, "pricing": { "prompt": "0.000002", "completion": "0.000012", "image": "0.000002", "audio": "0.000002", "input_audio_cache": "0.0000002", "web_search": "0.014", "internal_reasoning": "0.000012", "input_cache_read": "0.0000002", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000018", "audio": "0.000004", "input_audio_cache": "0.0000004", "input_cache_read": "0.0000004" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.91729128973896, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/llama-nemotron-embed-vl-1b-v2:free@nvidia", "name": "NVIDIA: Llama Nemotron Embed VL 1B V2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/llama-nemotron-embed-vl-1b-v2:free", "canonicalSlug": "nvidia/llama-nemotron-embed-vl-1b-v2-20260224", "servingProvider": "Nvidia", "servingProviderSlug": "nvidia", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "temperature", "max_tokens", "seed", "top_p" ], "architecture": { "modality": "text+image->embeddings", "input_modalities": [ "text", "image" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The Llama Nemotron Embed VL 1B V2 embedding model is optimized for multimodal question-answering retrieval. The model can embed 'documents' in the form of image, text, or image and text...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nvidia | nvidia/llama-nemotron-embed-vl-1b-v2-20260224:free", "model_id": "nvidia/llama-nemotron-embed-vl-1b-v2:free", "model_name": "NVIDIA: Llama Nemotron Embed VL 1B V2", "context_length": 131072, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Nvidia", "tag": "nvidia", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "temperature", "max_tokens", "seed", "top_p" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.3-codex@openai", "name": "OpenAI: GPT-5.3-Codex", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000175" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.3-codex", "canonicalSlug": "openai/gpt-5.3-codex-20260224", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.3-codex-20260224", "model_id": "openai/gpt-5.3-codex", "model_name": "OpenAI: GPT-5.3-Codex", "context_length": 400000, "pricing": { "prompt": "0.00000175", "completion": "0.000014", "web_search": "0.01", "input_cache_read": "0.000000175", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.98557605625338, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.3-codex@openai", "name": "OpenAI: GPT-5.3-Codex", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.3-codex", "canonicalSlug": "openai/gpt-5.3-codex-20260224", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.3-codex-20260224", "model_id": "openai/gpt-5.3-codex", "model_name": "OpenAI: GPT-5.3-Codex", "context_length": 400000, "pricing": { "prompt": "0.0000035", "completion": "0.000028", "web_search": "0.01", "input_cache_read": "0.00000035", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.98557605625338, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.3-codex@azure", "name": "OpenAI: GPT-5.3-Codex", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000175" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.3-codex", "canonicalSlug": "openai/gpt-5.3-codex-20260224", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.3-codex-20260224", "model_id": "openai/gpt-5.3-codex", "model_name": "OpenAI: GPT-5.3-Codex", "context_length": 400000, "pricing": { "prompt": "0.00000175", "completion": "0.000014", "web_search": "0.01", "input_cache_read": "0.000000175", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "aion-labs/aion-2.0@aionlabs", "name": "AionLabs: Aion-2.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "aion-labs/aion-2.0", "canonicalSlug": "aion-labs/aion-2.0-20260223", "servingProvider": "AionLabs", "servingProviderSlug": "aionlabs", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AionLabs | aion-labs/aion-2.0-20260223", "model_id": "aion-labs/aion-2.0", "model_name": "AionLabs: Aion-2.0", "context_length": 131072, "pricing": { "prompt": "0.0000008", "completion": "0.0000016", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "AionLabs", "tag": "aion-labs", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-pro-preview@google", "name": "Google: Gemini 3.1 Pro Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.000002, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000002" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-pro-preview", "canonicalSlug": "google/gemini-3.1-pro-preview-20260219", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "stop", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "audio", "file", "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...", "uptimeLast30m": 85.20641369803381, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-pro-preview-20260219", "model_id": "google/gemini-3.1-pro-preview", "model_name": "Google: Gemini 3.1 Pro Preview", "context_length": 1048576, "pricing": { "prompt": "0.000002", "completion": "0.000012", "image": "0.000002", "audio": "0.000002", "input_audio_cache": "0.0000002", "web_search": "0.014", "internal_reasoning": "0.000012", "input_cache_read": "0.0000002", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000018", "audio": "0.000004", "input_audio_cache": "0.0000004", "input_cache_read": "0.0000004" } ] }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "stop", "structured_outputs", "reasoning_effort" ], "status": -2, "uptime_last_30m": 85.20641369803381, "uptime_last_5m": 91.37199434229137, "uptime_last_1d": 95.44618431216897, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-pro-preview@google", "name": "Google: Gemini 3.1 Pro Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "other": [ { "amount": 0.000001, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000001" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-pro-preview", "canonicalSlug": "google/gemini-3.1-pro-preview-20260219", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "stop", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "audio", "file", "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...", "uptimeLast30m": 85.20641369803381, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-pro-preview-20260219", "model_id": "google/gemini-3.1-pro-preview", "model_name": "Google: Gemini 3.1 Pro Preview", "context_length": 1048576, "pricing": { "prompt": "0.000001", "completion": "0.000006", "image": "0.000001", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.000006", "input_cache_read": "0.0000001", "input_cache_write": "0.0000001875", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000002", "completion": "0.000009", "audio": "0.000002", "input_audio_cache": "0.0000002", "input_cache_read": "0.0000002" } ] }, "provider_name": "Google", "tag": "google-vertex/global/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "stop", "structured_outputs", "reasoning_effort" ], "status": -2, "uptime_last_30m": 85.20641369803381, "uptime_last_5m": 91.37199434229137, "uptime_last_1d": 95.44618431216897, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-pro-preview@google", "name": "Google: Gemini 3.1 Pro Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "output": [ { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000036" } ], "cacheWrite": [ { "amount": 0.675, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000675" } ], "other": [ { "amount": 0.0000036, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000036" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-pro-preview", "canonicalSlug": "google/gemini-3.1-pro-preview-20260219", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "stop", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "audio", "file", "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...", "uptimeLast30m": 85.20641369803381, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-pro-preview-20260219", "model_id": "google/gemini-3.1-pro-preview", "model_name": "Google: Gemini 3.1 Pro Preview", "context_length": 1048576, "pricing": { "prompt": "0.0000036", "completion": "0.0000216", "image": "0.0000036", "audio": "0.0000036", "input_audio_cache": "0.00000036", "web_search": "0.014", "internal_reasoning": "0.0000216", "input_cache_read": "0.00000036", "input_cache_write": "0.000000675", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000072", "completion": "0.0000324", "audio": "0.0000072", "input_audio_cache": "0.00000072", "input_cache_read": "0.00000072" } ] }, "provider_name": "Google", "tag": "google-vertex/global/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "stop", "structured_outputs", "reasoning_effort" ], "status": -2, "uptime_last_30m": 85.20641369803381, "uptime_last_5m": 91.37199434229137, "uptime_last_1d": 95.44618431216897, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-pro-preview@google-ai-studio", "name": "Google: Gemini 3.1 Pro Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.000002, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000002" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-pro-preview", "canonicalSlug": "google/gemini-3.1-pro-preview-20260219", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "audio", "file", "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...", "uptimeLast30m": 99.5473512978287, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-pro-preview-20260219", "model_id": "google/gemini-3.1-pro-preview", "model_name": "Google: Gemini 3.1 Pro Preview", "context_length": 1048576, "pricing": { "prompt": "0.000002", "completion": "0.000012", "image": "0.000002", "audio": "0.000002", "input_audio_cache": "0.0000002", "web_search": "0.014", "internal_reasoning": "0.000012", "input_cache_read": "0.0000002", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000004", "completion": "0.000018", "audio": "0.000004", "input_audio_cache": "0.0000004", "input_cache_read": "0.0000004" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.5473512978287, "uptime_last_5m": 99.78237214363439, "uptime_last_1d": 99.27358259137938, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-pro-preview@google-ai-studio", "name": "Google: Gemini 3.1 Pro Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "other": [ { "amount": 0.000001, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000001" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-pro-preview", "canonicalSlug": "google/gemini-3.1-pro-preview-20260219", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "audio", "file", "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...", "uptimeLast30m": 99.5473512978287, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-pro-preview-20260219", "model_id": "google/gemini-3.1-pro-preview", "model_name": "Google: Gemini 3.1 Pro Preview", "context_length": 1048576, "pricing": { "prompt": "0.000001", "completion": "0.000006", "image": "0.000001", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.000006", "input_cache_read": "0.0000001", "input_cache_write": "0.0000001875", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000002", "completion": "0.000009", "audio": "0.000002", "input_audio_cache": "0.0000002", "input_cache_read": "0.0000002" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.5473512978287, "uptime_last_5m": 99.78237214363439, "uptime_last_1d": 99.27358259137938, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-pro-preview@google-ai-studio", "name": "Google: Gemini 3.1 Pro Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "output": [ { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000036" } ], "cacheWrite": [ { "amount": 0.675, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000675" } ], "other": [ { "amount": 0.0000036, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000036" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-pro-preview", "canonicalSlug": "google/gemini-3.1-pro-preview-20260219", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "audio", "file", "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...", "uptimeLast30m": 99.5473512978287, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3.1-pro-preview-20260219", "model_id": "google/gemini-3.1-pro-preview", "model_name": "Google: Gemini 3.1 Pro Preview", "context_length": 1048576, "pricing": { "prompt": "0.0000036", "completion": "0.0000216", "image": "0.0000036", "audio": "0.0000036", "input_audio_cache": "0.00000036", "web_search": "0.014", "internal_reasoning": "0.0000216", "input_cache_read": "0.00000036", "input_cache_write": "0.000000675", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000072", "completion": "0.0000324", "audio": "0.0000072", "input_audio_cache": "0.00000072", "input_cache_read": "0.00000072" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.5473512978287, "uptime_last_5m": 99.78237214363439, "uptime_last_1d": 99.27358259137938, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3.1-pro-preview:batch@google", "name": "Google: Gemini 3.1 Pro Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000001, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000001" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3.1-pro-preview:batch", "canonicalSlug": "google/gemini-3.1-pro-preview-20260219", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "stop", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "audio", "file", "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3.1-pro-preview-20260219:batch", "model_id": "google/gemini-3.1-pro-preview:batch", "model_name": "Google: Gemini 3.1 Pro Preview", "context_length": 1048576, "pricing": { "prompt": "0.000001", "completion": "0.000006", "image": "0.000001", "audio": "0.000001", "web_search": "0.014", "internal_reasoning": "0.000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000002", "completion": "0.000009", "audio": "0.000002" } ] }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "stop", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.6@google", "name": "Anthropic: Claude Sonnet 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.6", "canonicalSlug": "anthropic/claude-4.6-sonnet-20260217", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.6-sonnet-20260217", "model_id": "anthropic/claude-sonnet-4.6", "model_name": "Anthropic: Claude Sonnet 4.6", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96450731448657, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.6@anthropic", "name": "Anthropic: Claude Sonnet 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.6", "canonicalSlug": "anthropic/claude-4.6-sonnet-20260217", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "uptimeLast30m": 99.87157534246576, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.6-sonnet-20260217", "model_id": "anthropic/claude-sonnet-4.6", "model_name": "Anthropic: Claude Sonnet 4.6", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.87157534246576, "uptime_last_5m": 99.13793103448276, "uptime_last_1d": 99.67133033350547, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.6@amazon-bedrock", "name": "Anthropic: Claude Sonnet 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.6", "canonicalSlug": "anthropic/claude-4.6-sonnet-20260217", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "uptimeLast30m": 99.939581602598, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.6-sonnet-20260217", "model_id": "anthropic/claude-sonnet-4.6", "model_name": "Anthropic: Claude Sonnet 4.6", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.939581602598, "uptime_last_5m": 99.94481236203092, "uptime_last_1d": 99.93429831880242, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.6@claude-platform-on-aws", "name": "Anthropic: Claude Sonnet 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.6", "canonicalSlug": "anthropic/claude-4.6-sonnet-20260217", "servingProvider": "Claude Platform on AWS", "servingProviderSlug": "claude-platform-on-aws", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "uptimeLast30m": 99.97986914947157, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Claude Platform on AWS | anthropic/claude-4.6-sonnet-20260217", "model_id": "anthropic/claude-sonnet-4.6", "model_name": "Anthropic: Claude Sonnet 4.6", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0 }, "provider_name": "Claude Platform on AWS", "tag": "claude-on-aws", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.97986914947157, "uptime_last_5m": 99.93377483443709, "uptime_last_1d": 99.78026862017506, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.6@azure", "name": "Anthropic: Claude Sonnet 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.6", "canonicalSlug": "anthropic/claude-4.6-sonnet-20260217", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "top_k", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-4.6-sonnet-20260217", "model_id": "anthropic/claude-sonnet-4.6", "model_name": "Anthropic: Claude Sonnet 4.6", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0 }, "provider_name": "Azure", "tag": "azure/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "top_k", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.77413890457369, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.6@google", "name": "Anthropic: Claude Sonnet 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000033" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.6", "canonicalSlug": "anthropic/claude-4.6-sonnet-20260217", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.6-sonnet-20260217", "model_id": "anthropic/claude-sonnet-4.6", "model_name": "Anthropic: Claude Sonnet 4.6", "context_length": 1000000, "pricing": { "prompt": "0.0000033", "completion": "0.0000165", "web_search": "0.01", "input_cache_read": "0.00000033", "input_cache_write": "0.000004125", "input_cache_write_1h": "0.0000066", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/europe", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.76038338658148, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.6@amazon-bedrock", "name": "Anthropic: Claude Sonnet 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000033" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.6", "canonicalSlug": "anthropic/claude-4.6-sonnet-20260217", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.6-sonnet-20260217", "model_id": "anthropic/claude-sonnet-4.6", "model_name": "Anthropic: Claude Sonnet 4.6", "context_length": 1000000, "pricing": { "prompt": "0.0000033", "completion": "0.0000165", "web_search": "0.01", "input_cache_read": "0.00000033", "input_cache_write": "0.000004125", "input_cache_write_1h": "0.0000066", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.6@amazon-bedrock", "name": "Anthropic: Claude Sonnet 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000033" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.6", "canonicalSlug": "anthropic/claude-4.6-sonnet-20260217", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.6-sonnet-20260217", "model_id": "anthropic/claude-sonnet-4.6", "model_name": "Anthropic: Claude Sonnet 4.6", "context_length": 1000000, "pricing": { "prompt": "0.0000033", "completion": "0.0000165", "web_search": "0.01", "input_cache_read": "0.00000033", "input_cache_write": "0.000004125", "input_cache_write_1h": "0.0000066", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/us", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.6@google", "name": "Anthropic: Claude Sonnet 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000033" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.6", "canonicalSlug": "anthropic/claude-4.6-sonnet-20260217", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.6-sonnet-20260217", "model_id": "anthropic/claude-sonnet-4.6", "model_name": "Anthropic: Claude Sonnet 4.6", "context_length": 1000000, "pricing": { "prompt": "0.0000033", "completion": "0.0000165", "web_search": "0.01", "input_cache_read": "0.00000033", "input_cache_write": "0.000004125", "input_cache_write_1h": "0.0000066", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us-east5", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.6:batch@anthropic", "name": "Anthropic: Claude Sonnet 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.6:batch", "canonicalSlug": "anthropic/claude-4.6-sonnet-20260217", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.6-sonnet-20260217:batch", "model_id": "anthropic/claude-sonnet-4.6:batch", "model_name": "Anthropic: Claude Sonnet 4.6", "context_length": 1000000, "pricing": { "prompt": "0.0000015", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.00000015", "input_cache_write": "0.000001875", "input_cache_write_1h": "0.000003", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-plus-02-15@alibaba", "name": "Qwen: Qwen3.5 Plus 2026-02-15", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "output": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-plus-02-15", "canonicalSlug": "qwen/qwen3.5-plus-20260216", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.5-plus-20260216", "model_id": "qwen/qwen3.5-plus-02-15", "model_name": "Qwen: Qwen3.5 Plus 2026-02-15", "context_length": 1000000, "pricing": { "prompt": "0.00000026", "completion": "0.00000156", "discount": 0, "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.000000325", "completion": "0.00000195" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 983616, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99728135279885, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-397b-a17b@digitalocean", "name": "Qwen: Qwen3.5 397B A17B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003025" } ], "output": [ { "amount": 1.9250000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001925" } ], "cacheRead": [ { "amount": 0.111, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000111" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-397b-a17b", "canonicalSlug": "qwen/qwen3.5-397b-a17b-20260216", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "structured_outputs", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | qwen/qwen3.5-397b-a17b-20260216", "model_id": "qwen/qwen3.5-397b-a17b", "model_name": "Qwen: Qwen3.5 397B A17B", "context_length": 131072, "pricing": { "prompt": "0.0000003025", "completion": "0.000001925", "input_cache_read": "0.000000111", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "structured_outputs", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 96.32279406108162, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-397b-a17b@alibaba", "name": "Qwen: Qwen3.5 397B A17B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000039" } ], "output": [ { "amount": 2.34, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000234" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-397b-a17b", "canonicalSlug": "qwen/qwen3.5-397b-a17b-20260216", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...", "uptimeLast30m": 99.94079336885731, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3.5-397b-a17b-20260216", "model_id": "qwen/qwen3.5-397b-a17b", "model_name": "Qwen: Qwen3.5 397B A17B", "context_length": 262144, "pricing": { "prompt": "0.00000039", "completion": "0.00000234", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 258048, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 99.94079336885731, "uptime_last_5m": 98.63013698630137, "uptime_last_1d": 99.8519136510585, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-397b-a17b@deepinfra", "name": "Qwen: Qwen3.5 397B A17B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-397b-a17b", "canonicalSlug": "qwen/qwen3.5-397b-a17b-20260216", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 81920, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3.5-397b-a17b-20260216", "model_id": "qwen/qwen3.5-397b-a17b", "model_name": "Qwen: Qwen3.5 397B A17B", "context_length": 262144, "pricing": { "prompt": "0.00000045", "completion": "0.000003", "input_cache_read": "0.00000022", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 81920, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 97.52547709133088, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-397b-a17b@parasail", "name": "Qwen: Qwen3.5 397B A17B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-397b-a17b", "canonicalSlug": "qwen/qwen3.5-397b-a17b-20260216", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...", "uptimeLast30m": 99.97368421052632, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | qwen/qwen3.5-397b-a17b-20260216", "model_id": "qwen/qwen3.5-397b-a17b", "model_name": "Qwen: Qwen3.5 397B A17B", "context_length": 262144, "pricing": { "prompt": "0.0000005", "completion": "0.0000036", "input_cache_read": "0.0000003", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.97368421052632, "uptime_last_5m": 100, "uptime_last_1d": 99.4719151422263, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-397b-a17b@atlascloud", "name": "Qwen: Qwen3.5 397B A17B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-397b-a17b", "canonicalSlug": "qwen/qwen3.5-397b-a17b-20260216", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...", "uptimeLast30m": 99.75476053087132, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | qwen/qwen3.5-397b-a17b-20260216", "model_id": "qwen/qwen3.5-397b-a17b", "model_name": "Qwen: Qwen3.5 397B A17B", "context_length": 262144, "pricing": { "prompt": "0.00000055", "completion": "0.0000035", "input_cache_read": "0.00000055", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.75476053087132, "uptime_last_5m": 100, "uptime_last_1d": 97.85035497020232, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-397b-a17b@phala", "name": "Qwen: Qwen3.5 397B A17B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.22499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000225" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-397b-a17b", "canonicalSlug": "qwen/qwen3.5-397b-a17b-20260216", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | qwen/qwen3.5-397b-a17b-20260216", "model_id": "qwen/qwen3.5-397b-a17b", "model_name": "Qwen: Qwen3.5 397B A17B", "context_length": 262144, "pricing": { "prompt": "0.00000055", "completion": "0.0000035", "input_cache_read": "0.000000225", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 25.135678391959797, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-397b-a17b@novita", "name": "Qwen: Qwen3.5 397B A17B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-397b-a17b", "canonicalSlug": "qwen/qwen3.5-397b-a17b-20260216", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...", "uptimeLast30m": 90.32258064516128, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3.5-397b-a17b-20260216", "model_id": "qwen/qwen3.5-397b-a17b", "model_name": "Qwen: Qwen3.5 397B A17B", "context_length": 262144, "pricing": { "prompt": "0.0000006", "completion": "0.0000036", "discount": 0 }, "provider_name": "Novita", "tag": "novita", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": -2, "uptime_last_30m": 90.32258064516128, "uptime_last_5m": null, "uptime_last_1d": 98.012376541945, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-397b-a17b@streamlake", "name": "Qwen: Qwen3.5 397B A17B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-397b-a17b", "canonicalSlug": "qwen/qwen3.5-397b-a17b-20260216", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 256000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...", "uptimeLast30m": 99.93339253996447, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | qwen/qwen3.5-397b-a17b-20260216", "model_id": "qwen/qwen3.5-397b-a17b", "model_name": "Qwen: Qwen3.5 397B A17B", "context_length": 256000, "pricing": { "prompt": "0.0000006", "completion": "0.0000036", "input_cache_read": "0.00000012", "discount": 0 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "status": 0, "uptime_last_30m": 99.93339253996447, "uptime_last_5m": 100, "uptime_last_1d": 97.73024065667558, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-397b-a17b@gmicloud", "name": "Qwen: Qwen3.5 397B A17B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-397b-a17b", "canonicalSlug": "qwen/qwen3.5-397b-a17b-20260216", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | qwen/qwen3.5-397b-a17b-20260216", "model_id": "qwen/qwen3.5-397b-a17b", "model_name": "Qwen: Qwen3.5 397B A17B", "context_length": 262144, "pricing": { "prompt": "0.0000006", "completion": "0.0000036", "discount": 0 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 0, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3.5-397b-a17b@venice", "name": "Qwen: Qwen3.5 397B A17B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3.5-397b-a17b", "canonicalSlug": "qwen/qwen3.5-397b-a17b-20260216", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 128000, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "text", "image", "video" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | qwen/qwen3.5-397b-a17b-20260216", "model_id": "qwen/qwen3.5-397b-a17b", "model_name": "Qwen: Qwen3.5 397B A17B", "context_length": 128000, "pricing": { "prompt": "0.00000075", "completion": "0.0000045", "discount": 0 }, "provider_name": "Venice", "tag": "venice", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 95.59597490631974, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.5@inceptron", "name": "MiniMax: MiniMax M2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.5", "canonicalSlug": "minimax/minimax-m2.5-20260211", "servingProvider": "Inceptron", "servingProviderSlug": "inceptron", "contextLength": 196608, "maxCompletionTokens": 196608, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "stop", "frequency_penalty", "logit_bias", "parallel_tool_calls", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Inceptron | minimax/minimax-m2.5-20260211", "model_id": "minimax/minimax-m2.5", "model_name": "MiniMax: MiniMax M2.5", "context_length": 196608, "pricing": { "prompt": "0.00000022", "completion": "0.0000009", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "Inceptron", "tag": "inceptron/fp8", "quantization": "fp8", "max_completion_tokens": 196608, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "top_p", "stop", "frequency_penalty", "logit_bias", "parallel_tool_calls", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.95788655285237, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.5@digitalocean", "name": "MiniMax: MiniMax M2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000225" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.5", "canonicalSlug": "minimax/minimax-m2.5-20260211", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "structured_outputs", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | minimax/minimax-m2.5-20260211", "model_id": "minimax/minimax-m2.5", "model_name": "MiniMax: MiniMax M2.5", "context_length": 65536, "pricing": { "prompt": "0.000000225", "completion": "0.0000009", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "structured_outputs", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96013381172759, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.5@venice", "name": "MiniMax: MiniMax M2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.5", "canonicalSlug": "minimax/minimax-m2.5-20260211", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 198000, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | minimax/minimax-m2.5-20260211", "model_id": "minimax/minimax-m2.5", "model_name": "MiniMax: MiniMax M2.5", "context_length": 198000, "pricing": { "prompt": "0.00000027", "completion": "0.00000095", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Venice", "tag": "venice", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.77942130530685, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.5@streamlake", "name": "MiniMax: MiniMax M2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 1.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000108" } ], "cacheRead": [ { "amount": 0.027, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000027" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.5", "canonicalSlug": "minimax/minimax-m2.5-20260211", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 200000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | minimax/minimax-m2.5-20260211", "model_id": "minimax/minimax-m2.5", "model_name": "MiniMax: MiniMax M2.5", "context_length": 200000, "pricing": { "prompt": "0.00000027", "completion": "0.00000108", "input_cache_read": "0.000000027", "discount": 0.1 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.80776325753898, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.5@atlascloud", "name": "MiniMax: MiniMax M2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.295, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000295" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.5", "canonicalSlug": "minimax/minimax-m2.5-20260211", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 196608, "maxCompletionTokens": 196608, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "structured_outputs", "response_format", "seed", "stop", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | minimax/minimax-m2.5-20260211", "model_id": "minimax/minimax-m2.5", "model_name": "MiniMax: MiniMax M2.5", "context_length": 196608, "pricing": { "prompt": "0.000000295", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 196608, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "structured_outputs", "response_format", "seed", "stop", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.86720197652872, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.5@friendli", "name": "MiniMax: MiniMax M2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.5", "canonicalSlug": "minimax/minimax-m2.5-20260211", "servingProvider": "Friendli", "servingProviderSlug": "friendli", "contextLength": 196608, "maxCompletionTokens": 196608, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Friendli | minimax/minimax-m2.5-20260211", "model_id": "minimax/minimax-m2.5", "model_name": "MiniMax: MiniMax M2.5", "context_length": 196608, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Friendli", "tag": "friendli", "quantization": "unknown", "max_completion_tokens": 196608, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.93446633754205, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.5@minimax", "name": "MiniMax: MiniMax M2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.5", "canonicalSlug": "minimax/minimax-m2.5-20260211", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tool_choice", "tools" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/minimax-m2.5-20260211", "model_id": "minimax/minimax-m2.5", "model_name": "MiniMax: MiniMax M2.5", "context_length": 204800, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.93072816918625, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.5@novita", "name": "MiniMax: MiniMax M2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.5", "canonicalSlug": "minimax/minimax-m2.5-20260211", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 204800, "maxCompletionTokens": 131100, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | minimax/minimax-m2.5-20260211", "model_id": "minimax/minimax-m2.5", "model_name": "MiniMax: MiniMax M2.5", "context_length": 204800, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 131100, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.94331065759637, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.5@siliconflow", "name": "MiniMax: MiniMax M2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.5", "canonicalSlug": "minimax/minimax-m2.5-20260211", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 196608, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | minimax/minimax-m2.5-20260211", "model_id": "minimax/minimax-m2.5", "model_name": "MiniMax: MiniMax M2.5", "context_length": 196608, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "max_tokens" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.5@minimax", "name": "MiniMax: MiniMax M2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.5", "canonicalSlug": "minimax/minimax-m2.5-20260211", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/minimax-m2.5-20260211", "model_id": "minimax/minimax-m2.5", "model_name": "MiniMax: MiniMax M2.5", "context_length": 204800, "pricing": { "prompt": "0.0000006", "completion": "0.0000024", "input_cache_read": "0.00000006", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax/highspeed", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.67548676984524, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5@streamlake", "name": "Z.ai: GLM 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 1.92, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000192" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5", "canonicalSlug": "z-ai/glm-5-20260211", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 198000, "maxCompletionTokens": 128000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "tool_choice", "tools", "max_tokens", "temperature", "top_p", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | z-ai/glm-5-20260211", "model_id": "z-ai/glm-5", "model_name": "Z.ai: GLM 5", "context_length": 198000, "pricing": { "prompt": "0.0000006", "completion": "0.00000192", "input_cache_read": "0.00000012", "discount": 0.4 }, "provider_name": "StreamLake", "tag": "streamlake/fp8", "quantization": "fp8", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "tool_choice", "tools", "max_tokens", "temperature", "top_p", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.83573565406581, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5@gmicloud", "name": "Z.ai: GLM 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 1.92, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000192" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5", "canonicalSlug": "z-ai/glm-5-20260211", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 202752, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | z-ai/glm-5-20260211", "model_id": "z-ai/glm-5", "model_name": "Z.ai: GLM 5", "context_length": 202752, "pricing": { "prompt": "0.0000006", "completion": "0.00000192", "input_cache_read": "0.00000012", "discount": 0.4 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.46483180428135, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5@deepinfra", "name": "Z.ai: GLM 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000208" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5", "canonicalSlug": "z-ai/glm-5-20260211", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 202752, "maxCompletionTokens": 16384, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "structured_outputs", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "uptimeLast30m": 99.79508196721312, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | z-ai/glm-5-20260211", "model_id": "z-ai/glm-5", "model_name": "Z.ai: GLM 5", "context_length": 202752, "pricing": { "prompt": "0.0000006", "completion": "0.00000208", "input_cache_read": "0.00000012", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "structured_outputs", "logit_bias" ], "status": 0, "uptime_last_30m": 99.79508196721312, "uptime_last_5m": 100, "uptime_last_1d": 99.78703362411436, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5@baidu", "name": "Z.ai: GLM 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007" } ], "output": [ { "amount": 2.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000224" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5", "canonicalSlug": "z-ai/glm-5-20260211", "servingProvider": "Baidu", "servingProviderSlug": "baidu", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "tool_choice", "tools", "response_format", "structured_outputs", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Baidu | z-ai/glm-5-20260211", "model_id": "z-ai/glm-5", "model_name": "Z.ai: GLM 5", "context_length": 202752, "pricing": { "prompt": "0.0000007", "completion": "0.00000224", "input_cache_read": "0.00000014", "discount": 0 }, "provider_name": "Baidu", "tag": "baidu/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "tool_choice", "tools", "response_format", "structured_outputs", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.73376409704149, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5@digitalocean", "name": "Z.ai: GLM 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5", "canonicalSlug": "z-ai/glm-5-20260211", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "structured_outputs", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "uptimeLast30m": 96.22641509433963, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | z-ai/glm-5-20260211", "model_id": "z-ai/glm-5", "model_name": "Z.ai: GLM 5", "context_length": 64000, "pricing": { "prompt": "0.00000075", "completion": "0.0000024", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "structured_outputs", "response_format", "tool_choice", "tools", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 96.22641509433963, "uptime_last_5m": 100, "uptime_last_1d": 94.06449518450384, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5@siliconflow", "name": "Z.ai: GLM 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 2.5500000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000255" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5", "canonicalSlug": "z-ai/glm-5-20260211", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | z-ai/glm-5-20260211", "model_id": "z-ai/glm-5", "model_name": "Z.ai: GLM 5", "context_length": 204800, "pricing": { "prompt": "0.00000095", "completion": "0.00000255", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.45700817471209, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5@atlascloud", "name": "Z.ai: GLM 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "output": [ { "amount": 3.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000315" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000019" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5", "canonicalSlug": "z-ai/glm-5-20260211", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 202752, "maxCompletionTokens": 202752, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "seed", "logit_bias", "stop", "min_p", "top_k", "frequency_penalty", "presence_penalty", "repetition_penalty", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | z-ai/glm-5-20260211", "model_id": "z-ai/glm-5", "model_name": "Z.ai: GLM 5", "context_length": 202752, "pricing": { "prompt": "0.00000095", "completion": "0.00000315", "input_cache_read": "0.00000019", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 202752, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "seed", "logit_bias", "stop", "min_p", "top_k", "frequency_penalty", "presence_penalty", "repetition_penalty", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 98.58483230522005, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5@amazon-bedrock", "name": "Z.ai: GLM 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5", "canonicalSlug": "z-ai/glm-5-20260211", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "tools", "tool_choice", "seed" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | z-ai/glm-5-20260211", "model_id": "z-ai/glm-5", "model_name": "Z.ai: GLM 5", "context_length": 202752, "pricing": { "prompt": "0.000001", "completion": "0.0000032", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "tools", "tool_choice", "seed" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 95.42682926829268, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5@novita", "name": "Z.ai: GLM 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5", "canonicalSlug": "z-ai/glm-5-20260211", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 202800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | z-ai/glm-5-20260211", "model_id": "z-ai/glm-5", "model_name": "Z.ai: GLM 5", "context_length": 202800, "pricing": { "prompt": "0.000001", "completion": "0.0000032", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96818471048087, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5@z.ai", "name": "Z.ai: GLM 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5", "canonicalSlug": "z-ai/glm-5-20260211", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-5-20260211", "model_id": "z-ai/glm-5", "model_name": "Z.ai: GLM 5", "context_length": 202752, "pricing": { "prompt": "0.000001", "completion": "0.0000032", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 98.52568101476399, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-5@venice", "name": "Z.ai: GLM 5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-5", "canonicalSlug": "z-ai/glm-5-20260211", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 198000, "maxCompletionTokens": 32000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | z-ai/glm-5-20260211", "model_id": "z-ai/glm-5", "model_name": "Z.ai: GLM 5", "context_length": 198000, "pricing": { "prompt": "0.000001", "completion": "0.0000032", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 95.55014605647517, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-max-thinking@alibaba", "name": "Qwen: Qwen3 Max Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.78, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000078" } ], "output": [ { "amount": 3.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000039" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-max-thinking", "canonicalSlug": "qwen/qwen3-max-thinking-20260123", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-max-thinking-20260123", "model_id": "qwen/qwen3-max-thinking", "model_name": "Qwen: Qwen3 Max Thinking", "context_length": 262144, "pricing": { "prompt": "0.00000078", "completion": "0.0000039", "discount": 0, "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.00000156", "completion": "0.0000078" }, { "min_prompt_tokens": 128000, "prompt": "0.00000195", "completion": "0.00000975" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 258048, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.6@amazon-bedrock", "name": "Anthropic: Claude Opus 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.6", "canonicalSlug": "anthropic/claude-4.6-opus-20260205", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...", "uptimeLast30m": 99.9282982791587, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.6-opus-20260205", "model_id": "anthropic/claude-opus-4.6", "model_name": "Anthropic: Claude Opus 4.6", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tool_choice", "tools", "response_format", "verbosity", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.9282982791587, "uptime_last_5m": 100, "uptime_last_1d": 99.79432066592723, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.6@azure", "name": "Anthropic: Claude Opus 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.6", "canonicalSlug": "anthropic/claude-4.6-opus-20260205", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "top_k", "stop", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-4.6-opus-20260205", "model_id": "anthropic/claude-opus-4.6", "model_name": "Anthropic: Claude Opus 4.6", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Azure", "tag": "azure/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity", "top_k", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.67637540453075, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.6@google", "name": "Anthropic: Claude Opus 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.6", "canonicalSlug": "anthropic/claude-4.6-opus-20260205", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.6-opus-20260205", "model_id": "anthropic/claude-opus-4.6", "model_name": "Anthropic: Claude Opus 4.6", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.76226116051774, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.6@anthropic", "name": "Anthropic: Claude Opus 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.6", "canonicalSlug": "anthropic/claude-4.6-opus-20260205", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.6-opus-20260205", "model_id": "anthropic/claude-opus-4.6", "model_name": "Anthropic: Claude Opus 4.6", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.58626284478096, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.6@claude-platform-on-aws", "name": "Anthropic: Claude Opus 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.6", "canonicalSlug": "anthropic/claude-4.6-opus-20260205", "servingProvider": "Claude Platform on AWS", "servingProviderSlug": "claude-platform-on-aws", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...", "uptimeLast30m": 99.92721979621543, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Claude Platform on AWS | anthropic/claude-4.6-opus-20260205", "model_id": "anthropic/claude-opus-4.6", "model_name": "Anthropic: Claude Opus 4.6", "context_length": 1000000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Claude Platform on AWS", "tag": "claude-on-aws", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.92721979621543, "uptime_last_5m": 99.82993197278913, "uptime_last_1d": 99.79098776395922, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.6@google", "name": "Anthropic: Claude Opus 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.6", "canonicalSlug": "anthropic/claude-4.6-opus-20260205", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.6-opus-20260205", "model_id": "anthropic/claude-opus-4.6", "model_name": "Anthropic: Claude Opus 4.6", "context_length": 1000000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/europe", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.6:batch@anthropic", "name": "Anthropic: Claude Opus 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [ { "amount": 3.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.6:batch", "canonicalSlug": "anthropic/claude-4.6-opus-20260205", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.6-opus-20260205:batch", "model_id": "anthropic/claude-opus-4.6:batch", "model_name": "Anthropic: Claude Opus 4.6", "context_length": 1000000, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "input_cache_write_1h": "0.000005", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder-next@parasail", "name": "Qwen: Qwen3 Coder Next", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder-next", "canonicalSlug": "qwen/qwen3-coder-next-2025-02-03", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | qwen/qwen3-coder-next-2025-02-03", "model_id": "qwen/qwen3-coder-next", "model_name": "Qwen: Qwen3 Coder Next", "context_length": 262144, "pricing": { "prompt": "0.00000012", "completion": "0.0000008", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.98093110681305, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder-next@streamlake", "name": "Qwen: Qwen3 Coder Next", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "cacheRead": [ { "amount": 0.036, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000036" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder-next", "canonicalSlug": "qwen/qwen3-coder-next-2025-02-03", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 256000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "tools", "tool_choice", "response_format", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...", "uptimeLast30m": 99.62168978562421, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | qwen/qwen3-coder-next-2025-02-03", "model_id": "qwen/qwen3-coder-next", "model_name": "Qwen: Qwen3 Coder Next", "context_length": 256000, "pricing": { "prompt": "0.00000018", "completion": "0.0000009", "input_cache_read": "0.000000036", "discount": 0.4 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "tools", "tool_choice", "response_format", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "status": 0, "uptime_last_30m": 99.62168978562421, "uptime_last_5m": 100, "uptime_last_1d": 99.55530216647662, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder-next@novita", "name": "Qwen: Qwen3 Coder Next", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder-next", "canonicalSlug": "qwen/qwen3-coder-next-2025-02-03", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...", "uptimeLast30m": 99.29078014184397, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3-coder-next-2025-02-03", "model_id": "qwen/qwen3-coder-next", "model_name": "Qwen: Qwen3 Coder Next", "context_length": 262144, "pricing": { "prompt": "0.0000002", "completion": "0.0000015", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": 99.29078014184397, "uptime_last_5m": 100, "uptime_last_1d": 99.62391876645356, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder-next@alibaba", "name": "Qwen: Qwen3 Coder Next", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder-next", "canonicalSlug": "qwen/qwen3-coder-next-2025-02-03", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-coder-next-2025-02-03", "model_id": "qwen/qwen3-coder-next", "model_name": "Qwen: Qwen3 Coder Next", "context_length": 262144, "pricing": { "prompt": "0.0000003", "completion": "0.0000015", "discount": 0, "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.0000005", "completion": "0.0000025" }, { "min_prompt_tokens": 128000, "prompt": "0.0000008", "completion": "0.000004" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 204800, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.32440732096795, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sourceful/riverflow-v2-pro@sourceful", "name": "Sourceful: Riverflow V2 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sourceful/riverflow-v2-pro", "canonicalSlug": "sourceful/riverflow-v2-pro-20260130", "servingProvider": "Sourceful", "servingProviderSlug": "sourceful", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Riverflow V2 Pro is the most powerful variant of Sourceful's Riverflow 2.0 lineup, best for top-tier control and perfect text rendering. The Riverflow 2.0 series represents SOTA performance on image...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Sourceful | sourceful/riverflow-v2-pro-20260130", "model_id": "sourceful/riverflow-v2-pro", "model_name": "Sourceful: Riverflow V2 Pro", "context_length": 8192, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.0000359281437125749", "image_output": "0.0000359281437125749", "discount": 0 }, "provider_name": "Sourceful", "tag": "sourceful", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sourceful/riverflow-v2-fast@sourceful", "name": "Sourceful: Riverflow V2 Fast", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sourceful/riverflow-v2-fast", "canonicalSlug": "sourceful/riverflow-v2-fast-20260130", "servingProvider": "Sourceful", "servingProviderSlug": "sourceful", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Riverflow V2 Fast is the fastest variant of Sourceful's Riverflow 2.0 lineup, best for production deployments and latency-critical workflows. The Riverflow 2.0 series represents SOTA performance on image generation and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Sourceful | sourceful/riverflow-v2-fast-20260130", "model_id": "sourceful/riverflow-v2-fast", "model_name": "Sourceful: Riverflow V2 Fast", "context_length": 8192, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.00000479041916167665", "image_output": "0.00000479041916167665", "discount": 0 }, "provider_name": "Sourceful", "tag": "sourceful", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openrouter/free", "name": "Free Models Router", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "canonicalSlug": "openrouter/free", "contextLength": 200000, "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "frequency_penalty", "include_reasoning", "logprobs", "max_completion_tokens", "max_tokens", "min_p", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_k", "top_logprobs", "top_p" ], "description": "The simplest way to get free inference. openrouter/free is a router that selects free models at random from the models available on OpenRouter. The router smartly filters for models that...", "endpointCount": 0 } }, { "id": "stepfun/step-3.5-flash@siliconflow", "name": "StepFun: Step 3.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "stepfun/step-3.5-flash", "canonicalSlug": "stepfun/step-3.5-flash", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | stepfun/step-3.5-flash", "model_id": "stepfun/step-3.5-flash", "model_name": "StepFun: Step 3.5 Flash", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.0000003", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.5@digitalocean", "name": "MoonshotAI: Kimi K2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "output": [ { "amount": 2.025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002025" } ], "cacheRead": [ { "amount": 0.203, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000203" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.5", "canonicalSlug": "moonshotai/kimi-k2.5-0127", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "structured_outputs", "tool_choice", "tools", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...", "uptimeLast30m": 99.53488372093024, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | moonshotai/kimi-k2.5-0127", "model_id": "moonshotai/kimi-k2.5", "model_name": "MoonshotAI: Kimi K2.5", "context_length": 262144, "pricing": { "prompt": "0.000000375", "completion": "0.000002025", "input_cache_read": "0.000000203", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "response_format", "structured_outputs", "tool_choice", "tools", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.53488372093024, "uptime_last_5m": 100, "uptime_last_1d": 99.48380104950947, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.5@deepinfra", "name": "MoonshotAI: Kimi K2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.5", "canonicalSlug": "moonshotai/kimi-k2.5-0127", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 64000, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...", "uptimeLast30m": 99.9625748502994, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | moonshotai/kimi-k2.5-0127", "model_id": "moonshotai/kimi-k2.5", "model_name": "MoonshotAI: Kimi K2.5", "context_length": 262144, "pricing": { "prompt": "0.00000045", "completion": "0.00000225", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias" ], "status": 0, "uptime_last_30m": 99.9625748502994, "uptime_last_5m": 100, "uptime_last_1d": 99.78123342677475, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.5@siliconflow", "name": "MoonshotAI: Kimi K2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.5", "canonicalSlug": "moonshotai/kimi-k2.5-0127", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "max_tokens" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...", "uptimeLast30m": 99.91452991452991, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | moonshotai/kimi-k2.5-0127", "model_id": "moonshotai/kimi-k2.5", "model_name": "MoonshotAI: Kimi K2.5", "context_length": 262144, "pricing": { "prompt": "0.00000045", "completion": "0.00000225", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/int4", "quantization": "int4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "max_tokens" ], "status": 0, "uptime_last_30m": 99.91452991452991, "uptime_last_5m": 100, "uptime_last_1d": 99.74111407053637, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.5@atlascloud", "name": "MoonshotAI: Kimi K2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.49, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000049" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.5", "canonicalSlug": "moonshotai/kimi-k2.5-0127", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | moonshotai/kimi-k2.5-0127", "model_id": "moonshotai/kimi-k2.5", "model_name": "MoonshotAI: Kimi K2.5", "context_length": 262144, "pricing": { "prompt": "0.00000049", "completion": "0.0000025", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/int4", "quantization": "int4", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 98.83375588554871, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.5@venice", "name": "MoonshotAI: Kimi K2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.532, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000532" } ], "output": [ { "amount": 3.3249999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003325" } ], "cacheRead": [ { "amount": 0.20900000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000209" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.5", "canonicalSlug": "moonshotai/kimi-k2.5-0127", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...", "uptimeLast30m": 99.68619246861925, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | moonshotai/kimi-k2.5-0127", "model_id": "moonshotai/kimi-k2.5", "model_name": "MoonshotAI: Kimi K2.5", "context_length": 256000, "pricing": { "prompt": "0.000000532", "completion": "0.000003325", "input_cache_read": "0.000000209", "discount": 0.05 }, "provider_name": "Venice", "tag": "venice", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.68619246861925, "uptime_last_5m": 100, "uptime_last_1d": 92.41013478904567, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.5@streamlake", "name": "MoonshotAI: Kimi K2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000054" } ], "output": [ { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "cacheRead": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.5", "canonicalSlug": "moonshotai/kimi-k2.5-0127", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 256000, "maxCompletionTokens": 256000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs", "structured_outputs", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | moonshotai/kimi-k2.5-0127", "model_id": "moonshotai/kimi-k2.5", "model_name": "MoonshotAI: Kimi K2.5", "context_length": 256000, "pricing": { "prompt": "0.00000054", "completion": "0.0000027", "input_cache_read": "0.00000009", "discount": 0.1 }, "provider_name": "StreamLake", "tag": "streamlake/fp8", "quantization": "fp8", "max_completion_tokens": 256000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs", "structured_outputs", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.89163831888511, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.5@novita", "name": "MoonshotAI: Kimi K2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5700000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000057" } ], "output": [ { "amount": 2.8499999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000285" } ], "cacheRead": [ { "amount": 0.095, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000095" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.5", "canonicalSlug": "moonshotai/kimi-k2.5-0127", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...", "uptimeLast30m": 99.27007299270073, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | moonshotai/kimi-k2.5-0127", "model_id": "moonshotai/kimi-k2.5", "model_name": "MoonshotAI: Kimi K2.5", "context_length": 262144, "pricing": { "prompt": "0.00000057", "completion": "0.00000285", "input_cache_read": "0.000000095", "discount": 0.05 }, "provider_name": "Novita", "tag": "novita", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.27007299270073, "uptime_last_5m": 98.09523809523809, "uptime_last_1d": 98.0138705697145, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.5@moonshot-ai", "name": "MoonshotAI: Kimi K2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.5", "canonicalSlug": "moonshotai/kimi-k2.5-0127", "servingProvider": "Moonshot AI", "servingProviderSlug": "moonshot-ai", "contextLength": 262144, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "frequency_penalty", "presence_penalty", "structured_outputs", "response_format", "tool_choice", "tools" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Moonshot AI | moonshotai/kimi-k2.5-0127", "model_id": "moonshotai/kimi-k2.5", "model_name": "MoonshotAI: Kimi K2.5", "context_length": 262144, "pricing": { "prompt": "0.0000006", "completion": "0.000003", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "Moonshot AI", "tag": "moonshotai/int4", "quantization": "int4", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "stop", "frequency_penalty", "presence_penalty", "structured_outputs", "response_format", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.97364619529145, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.5@amazon-bedrock", "name": "MoonshotAI: Kimi K2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.5", "canonicalSlug": "moonshotai/kimi-k2.5-0127", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | moonshotai/kimi-k2.5-0127", "model_id": "moonshotai/kimi-k2.5", "model_name": "MoonshotAI: Kimi K2.5", "context_length": 262144, "pricing": { "prompt": "0.0000006", "completion": "0.000003", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/us-east-2", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "seed", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 98.42562015206205, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2.5@phala", "name": "MoonshotAI: Kimi K2.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2.5", "canonicalSlug": "moonshotai/kimi-k2.5-0127", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tool_choice", "tools" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | moonshotai/kimi-k2.5-0127", "model_id": "moonshotai/kimi-k2.5", "model_name": "MoonshotAI: Kimi K2.5", "context_length": 262144, "pricing": { "prompt": "0.0000006", "completion": "0.000003", "input_cache_read": "0.00000022", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 80.33946251768033, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "upstage/solar-pro-3@upstage", "name": "Upstage: Solar Pro 3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "upstage/solar-pro-3", "canonicalSlug": "upstage/solar-pro-3", "servingProvider": "Upstage", "servingProviderSlug": "upstage", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "tools", "tool_choice", "structured_outputs", "response_format", "top_p", "frequency_penalty", "presence_penalty" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Upstage | upstage/solar-pro-3", "model_id": "upstage/solar-pro-3", "model_name": "Upstage: Solar Pro 3", "context_length": 131072, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.000000015", "discount": 0 }, "provider_name": "Upstage", "tag": "upstage", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "tools", "tool_choice", "structured_outputs", "response_format", "top_p", "frequency_penalty", "presence_penalty" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2-her@minimax", "name": "MiniMax: MiniMax M2-her", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2-her", "canonicalSlug": "minimax/minimax-m2-her-20260123", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 65536, "maxCompletionTokens": 2048, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/minimax-m2-her-20260123", "model_id": "minimax/minimax-m2-her", "model_name": "MiniMax: MiniMax M2-her", "context_length": 65536, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax", "quantization": "unknown", "max_completion_tokens": 2048, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "writer/palmyra-x5@amazon-bedrock", "name": "Writer: Palmyra X5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "writer/palmyra-x5", "canonicalSlug": "writer/palmyra-x5-20250428", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1040000, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | writer/palmyra-x5-20250428", "model_id": "writer/palmyra-x5", "model_name": "Writer: Palmyra X5", "context_length": 1040000, "pricing": { "prompt": "0.0000006", "completion": "0.000006", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-audio@openai", "name": "OpenAI: GPT Audio", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-audio", "canonicalSlug": "openai/gpt-audio", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+audio->text+audio", "input_modalities": [ "text", "audio" ], "output_modalities": [ "text", "audio" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-audio", "model_id": "openai/gpt-audio", "model_name": "OpenAI: GPT Audio", "context_length": 128000, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "audio": "0.000032", "audio_output": "0.000064", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-audio-mini@openai", "name": "OpenAI: GPT Audio Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000006" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-audio-mini", "canonicalSlug": "openai/gpt-audio-mini", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tool_choice", "tools" ], "architecture": { "modality": "text+audio->text+audio", "input_modalities": [ "text", "audio" ], "output_modalities": [ "text", "audio" ], "tokenizer": "GPT", "instruct_type": null }, "description": "A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-audio-mini", "model_id": "openai/gpt-audio-mini", "model_name": "OpenAI: GPT Audio Mini", "context_length": 128000, "pricing": { "prompt": "0.0000006", "completion": "0.0000024", "audio": "0.0000006", "audio_output": "0.0000024", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99615960674373, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7-flash@deepinfra", "name": "Z.ai: GLM 4.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7-flash", "canonicalSlug": "z-ai/glm-4.7-flash-20260119", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 202752, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "structured_outputs", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...", "uptimeLast30m": 99.89082969432314, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | z-ai/glm-4.7-flash-20260119", "model_id": "z-ai/glm-4.7-flash", "model_name": "Z.ai: GLM 4.7 Flash", "context_length": 202752, "pricing": { "prompt": "0.00000006", "completion": "0.0000004", "input_cache_read": "0.00000001", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "structured_outputs", "logit_bias" ], "status": 0, "uptime_last_30m": 99.89082969432314, "uptime_last_5m": 99.57983193277312, "uptime_last_1d": 99.59971850528302, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7-flash@venice", "name": "Z.ai: GLM 4.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7-flash", "canonicalSlug": "z-ai/glm-4.7-flash-20260119", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | z-ai/glm-4.7-flash-20260119", "model_id": "z-ai/glm-4.7-flash", "model_name": "Z.ai: GLM 4.7 Flash", "context_length": 128000, "pricing": { "prompt": "0.00000006", "completion": "0.0000004", "input_cache_read": "0.00000001", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.38996836873024, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7-flash@cloudflare", "name": "Z.ai: GLM 4.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.060500000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000605" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7-flash", "canonicalSlug": "z-ai/glm-4.7-flash-20260119", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "structured_outputs", "response_format", "min_p", "stop", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...", "uptimeLast30m": 99.96519317786287, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | z-ai/glm-4.7-flash-20260119", "model_id": "z-ai/glm-4.7-flash", "model_name": "Z.ai: GLM 4.7 Flash", "context_length": 131072, "pricing": { "prompt": "0.0000000605", "completion": "0.0000004", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "structured_outputs", "response_format", "min_p", "stop", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.96519317786287, "uptime_last_5m": 100, "uptime_last_1d": 98.4883396550369, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7-flash@novita", "name": "Z.ai: GLM 4.7 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7-flash", "canonicalSlug": "z-ai/glm-4.7-flash-20260119", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 200000, "maxCompletionTokens": 128000, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...", "uptimeLast30m": 96.82339990564554, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | z-ai/glm-4.7-flash-20260119", "model_id": "z-ai/glm-4.7-flash", "model_name": "Z.ai: GLM 4.7 Flash", "context_length": 200000, "pricing": { "prompt": "0.00000007", "completion": "0.0000004", "input_cache_read": "0.00000001", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": 96.82339990564554, "uptime_last_5m": 98.57369255150554, "uptime_last_1d": 92.84078606672288, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "black-forest-labs/flux.2-klein-4b@black-forest-labs", "name": "Black Forest Labs: FLUX.2 Klein 4B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "black-forest-labs/flux.2-klein-4b", "canonicalSlug": "black-forest-labs/flux.2-klein-4b", "servingProvider": "Black Forest Labs", "servingProviderSlug": "black-forest-labs", "contextLength": 40960, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "FLUX.2 [klein] 4B is the fastest and most cost-effective model in the FLUX.2 family, optimized for high-throughput use cases while maintaining excellent image quality. Pricing is based on the output...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Black Forest Labs | black-forest-labs/flux.2-klein-4b", "model_id": "black-forest-labs/flux.2-klein-4b", "model_name": "Black Forest Labs: FLUX.2 Klein 4B", "context_length": 40960, "pricing": { "prompt": "0", "completion": "0", "image_output": "0.00000341796875", "discount": 0 }, "provider_name": "Black Forest Labs", "tag": "black-forest-labs", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.2-codex@azure", "name": "OpenAI: GPT-5.2-Codex", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000175" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.2-codex", "canonicalSlug": "openai/gpt-5.2-codex-20260114", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.2-codex-20260114", "model_id": "openai/gpt-5.2-codex", "model_name": "OpenAI: GPT-5.2-Codex", "context_length": 400000, "pricing": { "prompt": "0.00000175", "completion": "0.000014", "web_search": "0.01", "input_cache_read": "0.000000175", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance-seed/seedream-4.5@seed", "name": "ByteDance Seed: Seedream 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance-seed/seedream-4.5", "canonicalSlug": "bytedance-seed/seedream-4.5-20251203", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "frequency_penalty", "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "Seedream 4.5 is the latest in-house image generation model developed by ByteDance. Compared with Seedream 4.0, it delivers comprehensive improvements, especially in editing consistency, including better preservation of subject details,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance-seed/seedream-4.5-20251203", "model_id": "bytedance-seed/seedream-4.5", "model_name": "ByteDance Seed: Seedream 4.5", "context_length": 4096, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.00000958083832335329", "image_output": "0.00000958083832335329", "discount": 0 }, "provider_name": "Seed", "tag": "seed", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "frequency_penalty", "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance-seed/seed-1.6-flash@seed", "name": "ByteDance Seed: Seed 1.6 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance-seed/seed-1.6-flash", "canonicalSlug": "bytedance-seed/seed-1.6-flash-20250625", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "structured_outputs", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance-seed/seed-1.6-flash-20250625", "model_id": "bytedance-seed/seed-1.6-flash", "model_name": "ByteDance Seed: Seed 1.6 Flash", "context_length": 262144, "pricing": { "prompt": "0.000000075", "completion": "0.0000003", "discount": 0, "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.0000001", "completion": "0.0000008" } ] }, "provider_name": "Seed", "tag": "seed/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "structured_outputs", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance-seed/seed-1.6@seed", "name": "ByteDance Seed: Seed 1.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance-seed/seed-1.6", "canonicalSlug": "bytedance-seed/seed-1.6-20250625", "servingProvider": "Seed", "servingProviderSlug": "seed", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "structured_outputs", "stop" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Seed | bytedance-seed/seed-1.6-20250625", "model_id": "bytedance-seed/seed-1.6", "model_name": "ByteDance Seed: Seed 1.6", "context_length": 262144, "pricing": { "prompt": "0.00000025", "completion": "0.000002", "discount": 0, "overrides": [ { "min_prompt_tokens": 128000, "prompt": "0.0000005", "completion": "0.000004" } ] }, "provider_name": "Seed", "tag": "seed/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "frequency_penalty", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format", "structured_outputs", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.1@novita", "name": "MiniMax: MiniMax M2.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.1", "canonicalSlug": "minimax/minimax-m2.1", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | minimax/minimax-m2.1", "model_id": "minimax/minimax-m2.1", "model_name": "MiniMax: MiniMax M2.1", "context_length": 204800, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.66672838363266, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.1@minimax", "name": "MiniMax: MiniMax M2.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.1", "canonicalSlug": "minimax/minimax-m2.1", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/minimax-m2.1", "model_id": "minimax/minimax-m2.1", "model_name": "MiniMax: MiniMax M2.1", "context_length": 204800, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.43253467843631, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2.1@minimax", "name": "MiniMax: MiniMax M2.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2.1", "canonicalSlug": "minimax/minimax-m2.1", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tool_choice", "tools" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/minimax-m2.1", "model_id": "minimax/minimax-m2.1", "model_name": "MiniMax: MiniMax M2.1", "context_length": 204800, "pricing": { "prompt": "0.0000003", "completion": "0.0000024", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax/highspeed", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.8868778280543, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7@deepinfra", "name": "Z.ai: GLM 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7", "canonicalSlug": "z-ai/glm-4.7-20251222", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "structured_outputs", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...", "uptimeLast30m": 99.86549278552214, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | z-ai/glm-4.7-20251222", "model_id": "z-ai/glm-4.7", "model_name": "Z.ai: GLM 4.7", "context_length": 202752, "pricing": { "prompt": "0.0000004", "completion": "0.00000175", "input_cache_read": "0.00000008", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "structured_outputs", "logit_bias" ], "status": 0, "uptime_last_30m": 99.86549278552214, "uptime_last_5m": 99.52830188679245, "uptime_last_1d": 99.3754999870971, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7@streamlake", "name": "Z.ai: GLM 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000048" } ], "output": [ { "amount": 1.76, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000176" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7", "canonicalSlug": "z-ai/glm-4.7-20251222", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 200000, "maxCompletionTokens": 128000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...", "uptimeLast30m": 95.13513513513514, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | z-ai/glm-4.7-20251222", "model_id": "z-ai/glm-4.7", "model_name": "Z.ai: GLM 4.7", "context_length": 200000, "pricing": { "prompt": "0.00000048", "completion": "0.00000176", "discount": 0.2 }, "provider_name": "StreamLake", "tag": "streamlake/fp8", "quantization": "fp8", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop" ], "status": 0, "uptime_last_30m": 95.13513513513514, "uptime_last_5m": 98.46153846153847, "uptime_last_1d": 93.12191448414359, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7@atlascloud", "name": "Z.ai: GLM 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000052" } ], "output": [ { "amount": 1.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000185" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7", "canonicalSlug": "z-ai/glm-4.7-20251222", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 202752, "maxCompletionTokens": 202752, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...", "uptimeLast30m": 96.81706890521161, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | z-ai/glm-4.7-20251222", "model_id": "z-ai/glm-4.7", "model_name": "Z.ai: GLM 4.7", "context_length": 202752, "pricing": { "prompt": "0.00000052", "completion": "0.00000185", "input_cache_read": "0.00000012", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 202752, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 96.81706890521161, "uptime_last_5m": 99.00990099009901, "uptime_last_1d": 97.45690636192667, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7@novita", "name": "Z.ai: GLM 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000054" } ], "output": [ { "amount": 1.9800000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000198" } ], "cacheRead": [ { "amount": 0.099, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000099" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7", "canonicalSlug": "z-ai/glm-4.7-20251222", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...", "uptimeLast30m": 95.49929676511954, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | z-ai/glm-4.7-20251222", "model_id": "z-ai/glm-4.7", "model_name": "Z.ai: GLM 4.7", "context_length": 204800, "pricing": { "prompt": "0.00000054", "completion": "0.00000198", "input_cache_read": "0.000000099", "discount": 0.1 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": 95.49929676511954, "uptime_last_5m": 97.6470588235294, "uptime_last_1d": 94.25893771426027, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7@venice", "name": "Z.ai: GLM 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "output": [ { "amount": 2.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000265" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7", "canonicalSlug": "z-ai/glm-4.7-20251222", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 198000, "maxCompletionTokens": 16384, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...", "uptimeLast30m": 99.71264367816092, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | z-ai/glm-4.7-20251222", "model_id": "z-ai/glm-4.7", "model_name": "Z.ai: GLM 4.7", "context_length": 198000, "pricing": { "prompt": "0.00000055", "completion": "0.00000265", "input_cache_read": "0.00000011", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp4", "quantization": "fp4", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 99.71264367816092, "uptime_last_5m": 99.48453608247422, "uptime_last_1d": 98.29256686018844, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7@z.ai", "name": "Z.ai: GLM 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7", "canonicalSlug": "z-ai/glm-4.7-20251222", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...", "uptimeLast30m": 95.26717557251908, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-4.7-20251222", "model_id": "z-ai/glm-4.7", "model_name": "Z.ai: GLM 4.7", "context_length": 202752, "pricing": { "prompt": "0.0000006", "completion": "0.0000022", "input_cache_read": "0.00000011", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai/fp4", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k", "response_format" ], "status": 0, "uptime_last_30m": 95.26717557251908, "uptime_last_5m": 97.67441860465115, "uptime_last_1d": 82.0658499536407, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7@google", "name": "Z.ai: GLM 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7", "canonicalSlug": "z-ai/glm-4.7-20251222", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 200000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...", "uptimeLast30m": 99.67672413793103, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | z-ai/glm-4.7-20251222", "model_id": "z-ai/glm-4.7", "model_name": "Z.ai: GLM 4.7", "context_length": 200000, "pricing": { "prompt": "0.0000006", "completion": "0.0000022", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.67672413793103, "uptime_last_5m": 100, "uptime_last_1d": 99.93217540031259, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.7@mancer-2", "name": "Z.ai: GLM 4.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.7", "canonicalSlug": "z-ai/glm-4.7-20251222", "servingProvider": "Mancer 2", "servingProviderSlug": "mancer-2", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...", "uptimeLast30m": 99.70501474926253, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mancer 2 | z-ai/glm-4.7-20251222", "model_id": "z-ai/glm-4.7", "model_name": "Z.ai: GLM 4.7", "context_length": 131072, "pricing": { "prompt": "0.0000006", "completion": "0.0000025", "discount": 0 }, "provider_name": "Mancer 2", "tag": "mancer/fp4", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a" ], "status": 0, "uptime_last_30m": 99.70501474926253, "uptime_last_5m": 100, "uptime_last_1d": 97.43938706588034, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3-flash-preview@google-ai-studio", "name": "Google: Gemini 3 Flash Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000005" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3-flash-preview", "canonicalSlug": "google/gemini-3-flash-preview-20251217", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...", "uptimeLast30m": 99.81610975344189, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3-flash-preview-20251217", "model_id": "google/gemini-3-flash-preview", "model_name": "Google: Gemini 3 Flash Preview", "context_length": 1048576, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "image": "0.0000005", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.000003", "input_cache_read": "0.00000005", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.81610975344189, "uptime_last_5m": 99.83863159593352, "uptime_last_1d": 99.70825076599434, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3-flash-preview@google-ai-studio", "name": "Google: Gemini 3 Flash Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 2.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000025" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3-flash-preview", "canonicalSlug": "google/gemini-3-flash-preview-20251217", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...", "uptimeLast30m": 99.81610975344189, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3-flash-preview-20251217", "model_id": "google/gemini-3-flash-preview", "model_name": "Google: Gemini 3 Flash Preview", "context_length": 1048576, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image": "0.00000025", "audio": "0.0000005", "input_audio_cache": "0.00000005", "web_search": "0.014", "internal_reasoning": "0.0000015", "input_cache_read": "0.000000025", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.81610975344189, "uptime_last_5m": 99.83863159593352, "uptime_last_1d": 99.70825076599434, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3-flash-preview@google-ai-studio", "name": "Google: Gemini 3 Flash Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "output": [ { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ], "cacheRead": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 9e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000009" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3-flash-preview", "canonicalSlug": "google/gemini-3-flash-preview-20251217", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...", "uptimeLast30m": 99.81610975344189, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3-flash-preview-20251217", "model_id": "google/gemini-3-flash-preview", "model_name": "Google: Gemini 3 Flash Preview", "context_length": 1048576, "pricing": { "prompt": "0.0000009", "completion": "0.0000054", "image": "0.0000009", "audio": "0.0000018", "input_audio_cache": "0.00000018", "web_search": "0.014", "internal_reasoning": "0.0000054", "input_cache_read": "0.00000009", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.81610975344189, "uptime_last_5m": 99.83863159593352, "uptime_last_1d": 99.70825076599434, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3-flash-preview@google", "name": "Google: Gemini 3 Flash Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000005" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3-flash-preview", "canonicalSlug": "google/gemini-3-flash-preview-20251217", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tool_choice", "tools", "stop", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...", "uptimeLast30m": 97.32491881624901, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3-flash-preview-20251217", "model_id": "google/gemini-3-flash-preview", "model_name": "Google: Gemini 3 Flash Preview", "context_length": 1048576, "pricing": { "prompt": "0.0000005", "completion": "0.000003", "image": "0.0000005", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.000003", "input_cache_read": "0.00000005", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tool_choice", "tools", "stop", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.32491881624901, "uptime_last_5m": 97.53452171729852, "uptime_last_1d": 91.94770884632658, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3-flash-preview@google", "name": "Google: Gemini 3 Flash Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 2.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000025" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3-flash-preview", "canonicalSlug": "google/gemini-3-flash-preview-20251217", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tool_choice", "tools", "stop", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...", "uptimeLast30m": 97.32491881624901, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3-flash-preview-20251217", "model_id": "google/gemini-3-flash-preview", "model_name": "Google: Gemini 3 Flash Preview", "context_length": 1048576, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image": "0.00000025", "audio": "0.0000005", "input_audio_cache": "0.00000005", "web_search": "0.014", "internal_reasoning": "0.0000015", "input_cache_read": "0.000000025", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tool_choice", "tools", "stop", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.32491881624901, "uptime_last_5m": 97.53452171729852, "uptime_last_1d": 91.94770884632658, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3-flash-preview@google", "name": "Google: Gemini 3 Flash Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "output": [ { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ], "cacheRead": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 9e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000009" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3-flash-preview", "canonicalSlug": "google/gemini-3-flash-preview-20251217", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tool_choice", "tools", "stop", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...", "uptimeLast30m": 97.32491881624901, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3-flash-preview-20251217", "model_id": "google/gemini-3-flash-preview", "model_name": "Google: Gemini 3 Flash Preview", "context_length": 1048576, "pricing": { "prompt": "0.0000009", "completion": "0.0000054", "image": "0.0000009", "audio": "0.0000018", "input_audio_cache": "0.00000018", "web_search": "0.014", "internal_reasoning": "0.0000054", "input_cache_read": "0.00000009", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tool_choice", "tools", "stop", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 97.32491881624901, "uptime_last_5m": 97.53452171729852, "uptime_last_1d": 91.94770884632658, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3-flash-preview:batch@google", "name": "Google: Gemini 3 Flash Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 2.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000025" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3-flash-preview:batch", "canonicalSlug": "google/gemini-3-flash-preview-20251217", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tool_choice", "tools", "stop", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-3-flash-preview-20251217:batch", "model_id": "google/gemini-3-flash-preview:batch", "model_name": "Google: Gemini 3 Flash Preview", "context_length": 1048576, "pricing": { "prompt": "0.00000025", "completion": "0.0000015", "image": "0.00000025", "audio": "0.0000005", "web_search": "0.014", "internal_reasoning": "0.0000015", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "tool_choice", "tools", "stop", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "black-forest-labs/flux.2-max@black-forest-labs", "name": "Black Forest Labs: FLUX.2 Max", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "black-forest-labs/flux.2-max", "canonicalSlug": "black-forest-labs/flux.2-max", "servingProvider": "Black Forest Labs", "servingProviderSlug": "black-forest-labs", "contextLength": 46864, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "FLUX.2 [max] is the new top-tier image model from Black Forest Labs, pushing image quality, prompt understanding, and editing consistency to the highest level yet. Pricing is as follows, [per...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Black Forest Labs | black-forest-labs/flux.2-max", "model_id": "black-forest-labs/flux.2-max", "model_name": "Black Forest Labs: FLUX.2 Max", "context_length": 46864, "pricing": { "prompt": "0", "completion": "0", "image_output": "0.00001708984375", "discount": 0 }, "provider_name": "Black Forest Labs", "tag": "black-forest-labs/us-3", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-nano-30b-a3b@crusoe", "name": "NVIDIA: Nemotron 3 Nano 30B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-nano-30b-a3b", "canonicalSlug": "nvidia/nemotron-3-nano-30b-a3b", "servingProvider": "Crusoe", "servingProviderSlug": "crusoe", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "structured_outputs", "tools", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Crusoe | nvidia/nemotron-3-nano-30b-a3b", "model_id": "nvidia/nemotron-3-nano-30b-a3b", "model_name": "NVIDIA: Nemotron 3 Nano 30B A3B", "context_length": 262144, "pricing": { "prompt": "0.00000005", "completion": "0.0000002", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Crusoe", "tag": "crusoe/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "structured_outputs", "tools", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99563161400941, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-nano-30b-a3b@deepinfra", "name": "NVIDIA: Nemotron 3 Nano 30B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-nano-30b-a3b", "canonicalSlug": "nvidia/nemotron-3-nano-30b-a3b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 228000, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...", "uptimeLast30m": 99.6551724137931, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | nvidia/nemotron-3-nano-30b-a3b", "model_id": "nvidia/nemotron-3-nano-30b-a3b", "model_name": "NVIDIA: Nemotron 3 Nano 30B A3B", "context_length": 262144, "pricing": { "prompt": "0.00000005", "completion": "0.0000002", "input_cache_read": "0.000000025", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 228000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "logit_bias" ], "status": 0, "uptime_last_30m": 99.6551724137931, "uptime_last_5m": null, "uptime_last_1d": 97.33148944888342, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-nano-30b-a3b@novita", "name": "NVIDIA: Nemotron 3 Nano 30B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-nano-30b-a3b", "canonicalSlug": "nvidia/nemotron-3-nano-30b-a3b", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | nvidia/nemotron-3-nano-30b-a3b", "model_id": "nvidia/nemotron-3-nano-30b-a3b", "model_name": "NVIDIA: Nemotron 3 Nano 30B A3B", "context_length": 262144, "pricing": { "prompt": "0.00000005", "completion": "0.0000002", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp4", "quantization": "fp4", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.68317358892439, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-3-nano-30b-a3b:free@nvidia", "name": "NVIDIA: Nemotron 3 Nano 30B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-3-nano-30b-a3b:free", "canonicalSlug": "nvidia/nemotron-3-nano-30b-a3b", "servingProvider": "Nvidia", "servingProviderSlug": "nvidia", "contextLength": 256000, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tool_choice", "tools" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...", "uptimeLast30m": 98.88220120378331, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nvidia | nvidia/nemotron-3-nano-30b-a3b:free", "model_id": "nvidia/nemotron-3-nano-30b-a3b:free", "model_name": "NVIDIA: Nemotron 3 Nano 30B A3B", "context_length": 256000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Nvidia", "tag": "nvidia/bf16", "quantization": "bf16", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 98.88220120378331, "uptime_last_5m": 99.13119026933101, "uptime_last_1d": 98.49086325405436, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.2-chat@azure", "name": "OpenAI: GPT-5.2 Chat", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000175" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.2-chat", "canonicalSlug": "openai/gpt-5.2-chat-20251211", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 128000, "maxCompletionTokens": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "structured_outputs", "response_format", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.2-chat-20251211", "model_id": "openai/gpt-5.2-chat", "model_name": "OpenAI: GPT-5.2 Chat", "context_length": 128000, "pricing": { "prompt": "0.00000175", "completion": "0.000014", "web_search": "0.01", "input_cache_read": "0.000000175", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 32000, "max_prompt_tokens": 96000, "supported_parameters": [ "max_completion_tokens", "structured_outputs", "response_format", "seed", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.2-pro@openai", "name": "OpenAI: GPT-5.2 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000021" } ], "output": [ { "amount": 168, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000168" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.2-pro", "canonicalSlug": "openai/gpt-5.2-pro-20251211", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.2-pro-20251211", "model_id": "openai/gpt-5.2-pro", "model_name": "OpenAI: GPT-5.2 Pro", "context_length": 400000, "pricing": { "prompt": "0.000021", "completion": "0.000168", "web_search": "0.01", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.2-pro:batch@openai", "name": "OpenAI: GPT-5.2 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000105" } ], "output": [ { "amount": 84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000084" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.2-pro:batch", "canonicalSlug": "openai/gpt-5.2-pro-20251211", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.2-pro-20251211:batch", "model_id": "openai/gpt-5.2-pro:batch", "model_name": "OpenAI: GPT-5.2 Pro", "context_length": 400000, "pricing": { "prompt": "0.0000105", "completion": "0.000084", "web_search": "0.01", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.2@openai", "name": "OpenAI: GPT-5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000175" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.2", "canonicalSlug": "openai/gpt-5.2-20251211", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...", "uptimeLast30m": 99.42096833805594, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.2-20251211", "model_id": "openai/gpt-5.2", "model_name": "OpenAI: GPT-5.2", "context_length": 400000, "pricing": { "prompt": "0.00000175", "completion": "0.000014", "web_search": "0.01", "input_cache_read": "0.000000175", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.42096833805594, "uptime_last_5m": 99.49579831932773, "uptime_last_1d": 99.41903940178315, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.2@openai", "name": "OpenAI: GPT-5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000875" } ], "output": [ { "amount": 7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000007" } ], "cacheRead": [ { "amount": 0.0875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000875" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.2", "canonicalSlug": "openai/gpt-5.2-20251211", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...", "uptimeLast30m": 99.42096833805594, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.2-20251211", "model_id": "openai/gpt-5.2", "model_name": "OpenAI: GPT-5.2", "context_length": 400000, "pricing": { "prompt": "0.000000875", "completion": "0.000007", "web_search": "0.01", "input_cache_read": "0.0000000875", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.42096833805594, "uptime_last_5m": 99.49579831932773, "uptime_last_1d": 99.41903940178315, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.2@openai", "name": "OpenAI: GPT-5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.2", "canonicalSlug": "openai/gpt-5.2-20251211", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...", "uptimeLast30m": 99.42096833805594, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.2-20251211", "model_id": "openai/gpt-5.2", "model_name": "OpenAI: GPT-5.2", "context_length": 400000, "pricing": { "prompt": "0.0000035", "completion": "0.000028", "web_search": "0.01", "input_cache_read": "0.00000035", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.42096833805594, "uptime_last_5m": 99.49579831932773, "uptime_last_1d": 99.41903940178315, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.2@azure", "name": "OpenAI: GPT-5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000175" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.2", "canonicalSlug": "openai/gpt-5.2-20251211", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.2-20251211", "model_id": "openai/gpt-5.2", "model_name": "OpenAI: GPT-5.2", "context_length": 400000, "pricing": { "prompt": "0.00000175", "completion": "0.000014", "web_search": "0.01", "input_cache_read": "0.000000175", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99413214411454, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.2:batch@openai", "name": "OpenAI: GPT-5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000875" } ], "output": [ { "amount": 7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000007" } ], "cacheRead": [ { "amount": 0.0875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000875" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.2:batch", "canonicalSlug": "openai/gpt-5.2-20251211", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.2-20251211:batch", "model_id": "openai/gpt-5.2:batch", "model_name": "OpenAI: GPT-5.2", "context_length": 400000, "pricing": { "prompt": "0.000000875", "completion": "0.000007", "web_search": "0.01", "input_cache_read": "0.0000000875", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.2:batch@openai", "name": "OpenAI: GPT-5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.4375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004375" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.04375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004375" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.2:batch", "canonicalSlug": "openai/gpt-5.2-20251211", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.2-20251211:batch", "model_id": "openai/gpt-5.2:batch", "model_name": "OpenAI: GPT-5.2", "context_length": 400000, "pricing": { "prompt": "0.0000004375", "completion": "0.0000035", "web_search": "0.01", "input_cache_read": "0.00000004375", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.2:batch@openai", "name": "OpenAI: GPT-5.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000175" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.2:batch", "canonicalSlug": "openai/gpt-5.2-20251211", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.2-20251211:batch", "model_id": "openai/gpt-5.2:batch", "model_name": "OpenAI: GPT-5.2", "context_length": 400000, "pricing": { "prompt": "0.00000175", "completion": "0.000014", "web_search": "0.01", "input_cache_read": "0.000000175", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "relace/relace-search@relace", "name": "Relace: Relace Search", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "relace/relace-search", "canonicalSlug": "relace/relace-search-20251208", "servingProvider": "Relace", "servingProviderSlug": "relace", "contextLength": 256000, "maxCompletionTokens": 128000, "quantization": "bf16", "status": 0, "supportedParameters": [ "tools", "tool_choice", "response_format", "temperature", "top_p", "stop", "seed", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Relace | relace/relace-search-20251208", "model_id": "relace/relace-search", "model_name": "Relace: Relace Search", "context_length": 256000, "pricing": { "prompt": "0.000001", "completion": "0.000003", "discount": 0 }, "provider_name": "Relace", "tag": "relace/bf16", "quantization": "bf16", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "tools", "tool_choice", "response_format", "temperature", "top_p", "stop", "seed", "max_tokens" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.6v@novita", "name": "Z.ai: GLM 4.6V", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "cacheRead": [ { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000055" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.6v", "canonicalSlug": "z-ai/glm-4.6-20251208", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "bf16", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...", "uptimeLast30m": 94.91017964071857, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | z-ai/glm-4.6-20251208", "model_id": "z-ai/glm-4.6v", "model_name": "Z.ai: GLM 4.6V", "context_length": 131072, "pricing": { "prompt": "0.0000003", "completion": "0.0000009", "input_cache_read": "0.000000055", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": -2, "uptime_last_30m": 94.91017964071857, "uptime_last_5m": 97.36842105263158, "uptime_last_1d": 89.84433843836305, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.6v@z.ai", "name": "Z.ai: GLM 4.6V", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.6v", "canonicalSlug": "z-ai/glm-4.6-20251208", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "fp8", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tool_choice", "tools", "top_k" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...", "uptimeLast30m": 94.2982456140351, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-4.6-20251208", "model_id": "z-ai/glm-4.6v", "model_name": "Z.ai: GLM 4.6V", "context_length": 131072, "pricing": { "prompt": "0.0000003", "completion": "0.0000009", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tool_choice", "tools", "top_k" ], "status": -2, "uptime_last_30m": 94.2982456140351, "uptime_last_5m": null, "uptime_last_1d": 79.89194268264036, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openrouter/bodybuilder", "name": "Body Builder (beta)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": -1000000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "-1" } ], "output": [ { "amount": -1000000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "-1" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "canonicalSlug": "openrouter/bodybuilder", "contextLength": 128000, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [], "description": "Transform your natural language requests into structured OpenRouter API request objects. Describe what you want to accomplish with AI models, and Body Builder will construct the appropriate API calls. Example:...", "endpointCount": 0 } }, { "id": "openai/gpt-5.1-codex-max@azure", "name": "OpenAI: GPT-5.1-Codex-Max", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.1-codex-max", "canonicalSlug": "openai/gpt-5.1-codex-max-20251204", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.1-codex-max-20251204", "model_id": "openai/gpt-5.1-codex-max", "model_name": "OpenAI: GPT-5.1-Codex-Max", "context_length": 400000, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.000000125", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "amazon/nova-2-lite-v1@amazon-bedrock", "name": "Amazon: Nova 2 Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "amazon/nova-2-lite-v1", "canonicalSlug": "amazon/nova-2-lite-v1", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tool_choice", "tools" ], "architecture": { "modality": "text+image+file+video->text", "input_modalities": [ "text", "image", "video", "file" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "description": "Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | amazon/nova-2-lite-v1", "model_id": "amazon/nova-2-lite-v1", "model_name": "Amazon: Nova 2 Lite", "context_length": 1000000, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.83337854070601, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "amazon/nova-2-lite-v1@amazon-bedrock", "name": "Amazon: Nova 2 Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "amazon/nova-2-lite-v1", "canonicalSlug": "amazon/nova-2-lite-v1", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tool_choice", "tools" ], "architecture": { "modality": "text+image+file+video->text", "input_modalities": [ "text", "image", "video", "file" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "description": "Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | amazon/nova-2-lite-v1", "model_id": "amazon/nova-2-lite-v1", "model_name": "Amazon: Nova 2 Lite", "context_length": 1000000, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96931887911639, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/ministral-14b-2512@mistral", "name": "Mistral: Ministral 3 14B 2512", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/ministral-14b-2512", "canonicalSlug": "mistralai/ministral-14b-2512", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...", "uptimeLast30m": 99.94093325457767, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/ministral-14b-2512", "model_id": "mistralai/ministral-14b-2512", "model_name": "Mistral: Ministral 3 14B 2512", "context_length": 262144, "pricing": { "prompt": "0.0000002", "completion": "0.0000002", "input_cache_read": "0.00000002", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.94093325457767, "uptime_last_5m": 99.48979591836735, "uptime_last_1d": 99.86958426250399, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/ministral-8b-2512@mistral", "name": "Mistral: Ministral 3 8B 2512", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/ministral-8b-2512", "canonicalSlug": "mistralai/ministral-8b-2512", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.", "uptimeLast30m": 99.94998749687421, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/ministral-8b-2512", "model_id": "mistralai/ministral-8b-2512", "model_name": "Mistral: Ministral 3 8B 2512", "context_length": 262144, "pricing": { "prompt": "0.00000015", "completion": "0.00000015", "input_cache_read": "0.000000015", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.94998749687421, "uptime_last_5m": 100, "uptime_last_1d": 99.89761389693673, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/ministral-3b-2512@mistral", "name": "Mistral: Ministral 3 3B 2512", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/ministral-3b-2512", "canonicalSlug": "mistralai/ministral-3b-2512", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tool_choice", "tools" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.", "uptimeLast30m": 99.95177236556547, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/ministral-3b-2512", "model_id": "mistralai/ministral-3b-2512", "model_name": "Mistral: Ministral 3 3B 2512", "context_length": 131072, "pricing": { "prompt": "0.0000001", "completion": "0.0000001", "input_cache_read": "0.00000001", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": 99.95177236556547, "uptime_last_5m": 100, "uptime_last_1d": 99.91620591858367, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-large-2512@mistral", "name": "Mistral: Mistral Large 3 2512", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-large-2512", "canonicalSlug": "mistralai/mistral-large-2512", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/mistral-large-2512", "model_id": "mistralai/mistral-large-2512", "model_name": "Mistral: Mistral Large 3 2512", "context_length": 262144, "pricing": { "prompt": "0.0000005", "completion": "0.0000015", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.94909407548813, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@gmicloud", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.20879999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002088" } ], "output": [ { "amount": 0.30960000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003096" } ], "cacheRead": [ { "amount": 0.0216, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000216" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "GMICloud", "servingProviderSlug": "gmicloud", "contextLength": 163840, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 99.64804223493181, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "GMICloud | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 163840, "pricing": { "prompt": "0.0000002088", "completion": "0.0000003096", "input_cache_read": "0.0000000216", "discount": 0.28 }, "provider_name": "GMICloud", "tag": "gmicloud/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.64804223493181, "uptime_last_5m": 99.47257383966245, "uptime_last_1d": 99.23124124673059, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@streamlake", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.2145, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002145" } ], "output": [ { "amount": 0.32175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000032175" } ], "cacheRead": [ { "amount": 0.02145, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002145" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 128000, "maxCompletionTokens": 64000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "max_tokens", "temperature", "top_p", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 99.90167901981546, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 128000, "pricing": { "prompt": "0.0000002145", "completion": "0.00000032175", "input_cache_read": "0.00000002145", "discount": 0.25 }, "provider_name": "StreamLake", "tag": "streamlake/fp8", "quantization": "fp8", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "structured_outputs", "logprobs", "top_logprobs", "tools", "tool_choice", "max_tokens", "temperature", "top_p", "stop" ], "status": 0, "uptime_last_30m": 99.90167901981546, "uptime_last_5m": 99.93342210386152, "uptime_last_1d": 99.84021542313644, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@digitalocean", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 163840, "maxCompletionTokens": 128000, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 93.71152685576342, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 163840, "pricing": { "prompt": "0.00000025", "completion": "0.0000008", "input_cache_read": "0.000000075", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "logprobs", "top_logprobs" ], "status": -2, "uptime_last_30m": 93.71152685576342, "uptime_last_5m": 99.77876106194691, "uptime_last_1d": 96.14510956707643, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@siliconflow", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25899999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000259" } ], "output": [ { "amount": 0.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000042" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000135" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 163840, "maxCompletionTokens": 163840, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 99.77551020408163, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 163840, "pricing": { "prompt": "0.000000259", "completion": "0.00000042", "input_cache_read": "0.000000135", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 163840, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "max_tokens" ], "status": 0, "uptime_last_30m": 99.77551020408163, "uptime_last_5m": 100, "uptime_last_1d": 99.69269025053956, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@deepinfra", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "output": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000038" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 163840, "maxCompletionTokens": 16384, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 98.6449864498645, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 163840, "pricing": { "prompt": "0.00000026", "completion": "0.00000038", "input_cache_read": "0.00000013", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tool_choice", "tools", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 98.6449864498645, "uptime_last_5m": 100, "uptime_last_1d": 95.63681534101937, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@atlascloud", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "output": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000038" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 163840, "maxCompletionTokens": 163840, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice", "seed", "logit_bias", "stop", "min_p", "top_k", "frequency_penalty", "presence_penalty", "repetition_penalty", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 99.96479431076062, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 163840, "pricing": { "prompt": "0.00000026", "completion": "0.00000038", "input_cache_read": "0.00000013", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 163840, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tools", "tool_choice", "seed", "logit_bias", "stop", "min_p", "top_k", "frequency_penalty", "presence_penalty", "repetition_penalty", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.96479431076062, "uptime_last_5m": 99.95115995115995, "uptime_last_1d": 99.93774793787676, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@novita", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.26899999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000269" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.13449999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001345" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 163840, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 99.98801246703428, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 163840, "pricing": { "prompt": "0.000000269", "completion": "0.0000004", "input_cache_read": "0.0000001345", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.98801246703428, "uptime_last_5m": 99.9683343888537, "uptime_last_1d": 99.98027847583619, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@baidu", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "output": [ { "amount": 0.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000042" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "Baidu", "servingProviderSlug": "baidu", "contextLength": 131072, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "stop", "seed", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 99.96137504828118, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Baidu | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 131072, "pricing": { "prompt": "0.00000028", "completion": "0.00000042", "input_cache_read": "0.000000028", "discount": 0 }, "provider_name": "Baidu", "tag": "baidu/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "stop", "seed", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.96137504828118, "uptime_last_5m": 100, "uptime_last_1d": 99.0827131413688, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@venice", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000033" } ], "output": [ { "amount": 0.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000048" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 160000, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "structured_outputs", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 160000, "pricing": { "prompt": "0.00000033", "completion": "0.00000048", "input_cache_read": "0.00000016", "discount": 0 }, "provider_name": "Venice", "tag": "venice", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "structured_outputs", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 92.89749798224375, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@alibaba", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3705, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003705" } ], "output": [ { "amount": 1.1115000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011115" } ], "cacheRead": [ { "amount": 0.0741, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000741" } ], "cacheWrite": [ { "amount": 0.46345, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000046345" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "structured_outputs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 97.524893314367, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 131072, "pricing": { "prompt": "0.0000003705", "completion": "0.0000011115", "input_cache_read": "0.0000000741", "input_cache_write": "0.00000046345", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": 98304, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "structured_outputs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 97.524893314367, "uptime_last_5m": 99.75308641975309, "uptime_last_1d": 98.14970166006971, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@friendli", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "Friendli", "servingProviderSlug": "friendli", "contextLength": 163840, "maxCompletionTokens": 163840, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 99.92106301802394, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Friendli | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 163840, "pricing": { "prompt": "0.0000005", "completion": "0.0000015", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "Friendli", "tag": "friendli", "quantization": "unknown", "max_completion_tokens": 163840, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "tools", "tool_choice", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.92106301802394, "uptime_last_5m": 100, "uptime_last_1d": 99.88119143239625, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@google", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000056" } ], "output": [ { "amount": 1.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000168" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 163840, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 99.50657894736842, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 163840, "pricing": { "prompt": "0.00000056", "completion": "0.00000168", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.50657894736842, "uptime_last_5m": 100, "uptime_last_1d": 99.58458119988407, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@phala", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 163840, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "structured_outputs", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 98.63429438543247, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 163840, "pricing": { "prompt": "0.000001", "completion": "0.000001", "input_cache_read": "0.0000005", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "structured_outputs", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": 98.63429438543247, "uptime_last_5m": 100, "uptime_last_1d": 99.27577265818258, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2@sambanova", "name": "DeepSeek: DeepSeek V3.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2", "canonicalSlug": "deepseek/deepseek-v3.2-20251201", "servingProvider": "SambaNova", "servingProviderSlug": "sambanova", "contextLength": 32768, "maxCompletionTokens": 7168, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SambaNova | deepseek/deepseek-v3.2-20251201", "model_id": "deepseek/deepseek-v3.2", "model_name": "DeepSeek: DeepSeek V3.2", "context_length": 32768, "pricing": { "prompt": "0.000003", "completion": "0.0000045", "discount": 0 }, "provider_name": "SambaNova", "tag": "sambanova", "quantization": "unknown", "max_completion_tokens": 7168, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 90.28511087645195, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "black-forest-labs/flux.2-flex@black-forest-labs", "name": "Black Forest Labs: FLUX.2 Flex", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "black-forest-labs/flux.2-flex", "canonicalSlug": "black-forest-labs/flux.2-flex", "servingProvider": "Black Forest Labs", "servingProviderSlug": "black-forest-labs", "contextLength": 67344, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "FLUX.2 [flex] excels at rendering complex text, typography, and fine details, and supports multi-reference editing in the same unified architecture. Pricing is as follows, [per the docs](https://bfl.ai/pricing?category=flux.2): We charge $0.06...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Black Forest Labs | black-forest-labs/flux.2-flex", "model_id": "black-forest-labs/flux.2-flex", "model_name": "Black Forest Labs: FLUX.2 Flex", "context_length": 67344, "pricing": { "prompt": "0", "completion": "0", "image_token": "0.0000146484375", "image_output": "0.0000146484375", "discount": 0 }, "provider_name": "Black Forest Labs", "tag": "black-forest-labs/us-3", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "black-forest-labs/flux.2-pro@black-forest-labs", "name": "Black Forest Labs: FLUX.2 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "black-forest-labs/flux.2-pro", "canonicalSlug": "black-forest-labs/flux.2-pro", "servingProvider": "Black Forest Labs", "servingProviderSlug": "black-forest-labs", "contextLength": 46864, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed" ], "architecture": { "modality": "text+image->image", "input_modalities": [ "text", "image" ], "output_modalities": [ "image" ], "tokenizer": "Other", "instruct_type": null }, "description": "A high-end image generation and editing model focused on frontier-level visual quality and reliability. It delivers strong prompt adherence, stable lighting, sharp textures, and consistent character/style reproduction across multi-reference inputs....", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Black Forest Labs | black-forest-labs/flux.2-pro", "model_id": "black-forest-labs/flux.2-pro", "model_name": "Black Forest Labs: FLUX.2 Pro", "context_length": 46864, "pricing": { "prompt": "0", "completion": "0", "image_output": "0.00000732421875", "discount": 0 }, "provider_name": "Black Forest Labs", "tag": "black-forest-labs", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.5@anthropic", "name": "Anthropic: Claude Opus 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.5", "canonicalSlug": "anthropic/claude-4.5-opus-20251124", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.5-opus-20251124", "model_id": "anthropic/claude-opus-4.5", "model_name": "Anthropic: Claude Opus 4.5", "context_length": 200000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.57645065650148, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.5@amazon-bedrock", "name": "Anthropic: Claude Opus 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.5", "canonicalSlug": "anthropic/claude-4.5-opus-20251124", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_k", "stop", "tool_choice", "tools", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...", "uptimeLast30m": 99.8546511627907, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.5-opus-20251124", "model_id": "anthropic/claude-opus-4.5", "model_name": "Anthropic: Claude Opus 4.5", "context_length": 200000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_k", "stop", "tool_choice", "tools", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 99.8546511627907, "uptime_last_5m": 100, "uptime_last_1d": 99.75483429227728, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.5@google", "name": "Anthropic: Claude Opus 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.5", "canonicalSlug": "anthropic/claude-4.5-opus-20251124", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "verbosity" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.5-opus-20251124", "model_id": "anthropic/claude-opus-4.5", "model_name": "Anthropic: Claude Opus 4.5", "context_length": 200000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "verbosity" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.62053571428572, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.5@azure", "name": "Anthropic: Claude Opus 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.5", "canonicalSlug": "anthropic/claude-4.5-opus-20251124", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-4.5-opus-20251124", "model_id": "anthropic/claude-opus-4.5", "model_name": "Anthropic: Claude Opus 4.5", "context_length": 200000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Azure", "tag": "azure/global", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "max_completion_tokens", "tools", "tool_choice", "response_format", "verbosity" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.5@claude-platform-on-aws", "name": "Anthropic: Claude Opus 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.5", "canonicalSlug": "anthropic/claude-4.5-opus-20251124", "servingProvider": "Claude Platform on AWS", "servingProviderSlug": "claude-platform-on-aws", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Claude Platform on AWS | anthropic/claude-4.5-opus-20251124", "model_id": "anthropic/claude-opus-4.5", "model_name": "Anthropic: Claude Opus 4.5", "context_length": 200000, "pricing": { "prompt": "0.000005", "completion": "0.000025", "web_search": "0.01", "input_cache_read": "0.0000005", "input_cache_write": "0.00000625", "input_cache_write_1h": "0.00001", "discount": 0 }, "provider_name": "Claude Platform on AWS", "tag": "claude-on-aws", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 97.96747967479675, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.5@amazon-bedrock", "name": "Anthropic: Claude Opus 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.5", "canonicalSlug": "anthropic/claude-4.5-opus-20251124", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_k", "stop", "tool_choice", "tools", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.5-opus-20251124", "model_id": "anthropic/claude-opus-4.5", "model_name": "Anthropic: Claude Opus 4.5", "context_length": 200000, "pricing": { "prompt": "0.0000055", "completion": "0.0000275", "web_search": "0.01", "input_cache_read": "0.00000055", "input_cache_write": "0.000006875", "input_cache_write_1h": "0.000011", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_k", "stop", "tool_choice", "tools", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.5:batch@anthropic", "name": "Anthropic: Claude Opus 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [ { "amount": 3.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.5:batch", "canonicalSlug": "anthropic/claude-4.5-opus-20251124", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.5-opus-20251124:batch", "model_id": "anthropic/claude-opus-4.5:batch", "model_name": "Anthropic: Claude Opus 4.5", "context_length": 200000, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "web_search": "0.01", "input_cache_read": "0.00000025", "input_cache_write": "0.000003125", "input_cache_write_1h": "0.000005", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "response_format", "verbosity" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "allenai/olmo-3-32b-think", "name": "AllenAI: Olmo 3 32B Think", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "canonicalSlug": "allenai/olmo-3-32b-think-20251121", "contextLength": 65536, "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "supportedParameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "max_tokens", "presence_penalty", "reasoning", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "top_k", "top_p" ], "description": "Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...", "endpointCount": 0 } }, { "id": "google/gemini-3-pro-image-preview@google-ai-studio", "name": "Google: Nano Banana Pro (Gemini 3 Pro Image Preview)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000002" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000000375" } ], "other": [ { "amount": 0.000002, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000002" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3-pro-image-preview", "canonicalSlug": "google/gemini-3-pro-image-preview-20251120", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 65536, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3-pro-image-preview-20251120", "model_id": "google/gemini-3-pro-image-preview", "model_name": "Google: Nano Banana Pro (Gemini 3 Pro Image Preview)", "context_length": 65536, "pricing": { "prompt": "0.000002", "completion": "0.000012", "image": "0.000002", "image_output": "0.00012", "audio": "0.000002", "input_audio_cache": "0.0000002", "web_search": "0.014", "internal_reasoning": "0.000012", "input_cache_read": "0.0000002", "input_cache_write": "0.000000375", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/global", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99858366971178, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-3-pro-image-preview@google-ai-studio", "name": "Google: Nano Banana Pro (Gemini 3 Pro Image Preview)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000001" } ], "cacheWrite": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000001875" } ], "other": [ { "amount": 0.000001, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000001" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000006" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-3-pro-image-preview", "canonicalSlug": "google/gemini-3-pro-image-preview-20251120", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 65536, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-3-pro-image-preview-20251120", "model_id": "google/gemini-3-pro-image-preview", "model_name": "Google: Nano Banana Pro (Gemini 3 Pro Image Preview)", "context_length": 65536, "pricing": { "prompt": "0.000001", "completion": "0.000006", "image": "0.000001", "image_output": "0.00006", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.000006", "input_cache_read": "0.0000001", "input_cache_write": "0.0000001875", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/global/flex", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99858366971178, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thenlper/gte-base@deepinfra", "name": "Thenlper: GTE-Base", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.000000005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thenlper/gte-base", "canonicalSlug": "thenlper/gte-base-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The gte-base embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, delivering efficient and effective semantic embeddings optimized for textual similarity, semantic search, and clustering applications.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | thenlper/gte-base-20251117", "model_id": "thenlper/gte-base", "model_name": "Thenlper: GTE-Base", "context_length": 512, "pricing": { "prompt": "0.000000005", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thenlper/gte-large@deepinfra", "name": "Thenlper: GTE-Large", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thenlper/gte-large", "canonicalSlug": "thenlper/gte-large-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The gte-large embedding model converts English sentences, paragraphs and moderate-length documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for information retrieval, semantic textual similarity, reranking and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | thenlper/gte-large-20251117", "model_id": "thenlper/gte-large", "model_name": "Thenlper: GTE-Large", "context_length": 512, "pricing": { "prompt": "0.00000001", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "intfloat/e5-large-v2@deepinfra", "name": "Intfloat: E5-Large-v2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "intfloat/e5-large-v2", "canonicalSlug": "intfloat/e5-large-v2-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The e5-large-v2 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-accuracy semantic embeddings optimized for retrieval, semantic search, reranking, and similarity-scoring tasks.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | intfloat/e5-large-v2-20251117", "model_id": "intfloat/e5-large-v2", "model_name": "Intfloat: E5-Large-v2", "context_length": 512, "pricing": { "prompt": "0.00000001", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "intfloat/e5-base-v2@deepinfra", "name": "Intfloat: E5-Base-v2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.000000005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "intfloat/e5-base-v2", "canonicalSlug": "intfloat/e5-base-v2-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The e5-base-v2 embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, similarity scoring,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | intfloat/e5-base-v2-20251117", "model_id": "intfloat/e5-base-v2", "model_name": "Intfloat: E5-Base-v2", "context_length": 512, "pricing": { "prompt": "0.000000005", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "intfloat/multilingual-e5-large@deepinfra", "name": "Intfloat: Multilingual-E5-Large", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "intfloat/multilingual-e5-large", "canonicalSlug": "intfloat/multilingual-e5-large-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "fp32", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The multilingual-e5-large embedding model encodes sentences, paragraphs, and documents across over 90 languages into a 1024-dimensional dense vector space, delivering robust semantic embeddings optimized for multilingual retrieval, cross-language similarity, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | intfloat/multilingual-e5-large-20251117", "model_id": "intfloat/multilingual-e5-large", "model_name": "Intfloat: Multilingual-E5-Large", "context_length": 512, "pricing": { "prompt": "0.00000001", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp32", "quantization": "fp32", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sentence-transformers/paraphrase-minilm-l6-v2@deepinfra", "name": "Sentence Transformers: paraphrase-MiniLM-L6-v2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.000000005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sentence-transformers/paraphrase-minilm-l6-v2", "canonicalSlug": "sentence-transformers/paraphrase-minilm-l6-v2-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The paraphrase-MiniLM-L6-v2 embedding model converts sentences and short paragraphs into a 384-dimensional dense vector space, producing high-quality semantic embeddings optimized for paraphrase detection, semantic similarity scoring, clustering, and lightweight retrieval...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | sentence-transformers/paraphrase-minilm-l6-v2-20251117", "model_id": "sentence-transformers/paraphrase-minilm-l6-v2", "model_name": "Sentence Transformers: paraphrase-MiniLM-L6-v2", "context_length": 512, "pricing": { "prompt": "0.000000005", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sentence-transformers/all-minilm-l12-v2@deepinfra", "name": "Sentence Transformers: all-MiniLM-L12-v2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.000000005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sentence-transformers/all-minilm-l12-v2", "canonicalSlug": "sentence-transformers/all-minilm-l12-v2-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The all-MiniLM-L12-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, clustering, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | sentence-transformers/all-minilm-l12-v2-20251117", "model_id": "sentence-transformers/all-minilm-l12-v2", "model_name": "Sentence Transformers: all-MiniLM-L12-v2", "context_length": 512, "pricing": { "prompt": "0.000000005", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "baai/bge-base-en-v1.5@deepinfra", "name": "BAAI: bge-base-en-v1.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.000000005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "baai/bge-base-en-v1.5", "canonicalSlug": "baai/bge-base-en-v1.5-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The bge-base-en-v1.5 embedding model converts English sentences and paragraphs into 768-dimensional dense vectors, delivering efficient, high-quality semantic embeddings optimized for retrieval, semantic search, and document-matching workflows. This version (v1.5) features...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | baai/bge-base-en-v1.5-20251117", "model_id": "baai/bge-base-en-v1.5", "model_name": "BAAI: bge-base-en-v1.5", "context_length": 512, "pricing": { "prompt": "0.000000005", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sentence-transformers/multi-qa-mpnet-base-dot-v1@deepinfra", "name": "Sentence Transformers: multi-qa-mpnet-base-dot-v1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.000000005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sentence-transformers/multi-qa-mpnet-base-dot-v1", "canonicalSlug": "sentence-transformers/multi-qa-mpnet-base-dot-v1-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The multi-qa-mpnet-base-dot-v1 embedding model transforms sentences and short paragraphs into a 768-dimensional dense vector space, generating high-quality semantic embeddings optimized for question-and-answer retrieval, semantic search, and similarity-scoring across diverse content.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | sentence-transformers/multi-qa-mpnet-base-dot-v1-20251117", "model_id": "sentence-transformers/multi-qa-mpnet-base-dot-v1", "model_name": "Sentence Transformers: multi-qa-mpnet-base-dot-v1", "context_length": 512, "pricing": { "prompt": "0.000000005", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "baai/bge-large-en-v1.5@deepinfra", "name": "BAAI: bge-large-en-v1.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "baai/bge-large-en-v1.5", "canonicalSlug": "baai/bge-large-en-v1.5-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The bge-large-en-v1.5 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-fidelity semantic embeddings optimized for semantic search, document retrieval, and downstream NLP tasks...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | baai/bge-large-en-v1.5-20251117", "model_id": "baai/bge-large-en-v1.5", "model_name": "BAAI: bge-large-en-v1.5", "context_length": 512, "pricing": { "prompt": "0.00000001", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "baai/bge-m3@deepinfra", "name": "BAAI: bge-m3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "baai/bge-m3", "canonicalSlug": "baai/bge-m3-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 8192, "quantization": "fp32", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The bge-m3 embedding model encodes sentences, paragraphs, and long documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for multilingual retrieval, semantic search, and large-context applications.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | baai/bge-m3-20251117", "model_id": "baai/bge-m3", "model_name": "BAAI: bge-m3", "context_length": 8192, "pricing": { "prompt": "0.00000001", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp32", "quantization": "fp32", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99625226289996, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "baai/bge-m3@parasail", "name": "BAAI: bge-m3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "baai/bge-m3", "canonicalSlug": "baai/bge-m3-20251117", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 8194, "maxCompletionTokens": 8194, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The bge-m3 embedding model encodes sentences, paragraphs, and long documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for multilingual retrieval, semantic search, and large-context applications.", "uptimeLast30m": 99.98595604241275, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | baai/bge-m3-20251117", "model_id": "baai/bge-m3", "model_name": "BAAI: bge-m3", "context_length": 8194, "pricing": { "prompt": "0.00000001", "completion": "0", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail", "quantization": "unknown", "max_completion_tokens": 8194, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias" ], "status": 0, "uptime_last_30m": 99.98595604241275, "uptime_last_5m": 100, "uptime_last_1d": 99.99551159483295, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sentence-transformers/all-mpnet-base-v2@deepinfra", "name": "Sentence Transformers: all-mpnet-base-v2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.000000005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sentence-transformers/all-mpnet-base-v2", "canonicalSlug": "sentence-transformers/all-mpnet-base-v2-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The all-mpnet-base-v2 embedding model encodes sentences and short paragraphs into a 768-dimensional dense vector space, providing high-fidelity semantic embeddings well suited for tasks like information retrieval, clustering, similarity scoring, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | sentence-transformers/all-mpnet-base-v2-20251117", "model_id": "sentence-transformers/all-mpnet-base-v2", "model_name": "Sentence Transformers: all-mpnet-base-v2", "context_length": 512, "pricing": { "prompt": "0.000000005", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sentence-transformers/all-minilm-l6-v2@deepinfra", "name": "Sentence Transformers: all-MiniLM-L6-v2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.000000005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sentence-transformers/all-minilm-l6-v2", "canonicalSlug": "sentence-transformers/all-minilm-l6-v2-20251117", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 512, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The all-MiniLM-L6-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, enabling high-quality semantic representations that are ideal for downstream tasks such as information retrieval, clustering,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | sentence-transformers/all-minilm-l6-v2-20251117", "model_id": "sentence-transformers/all-minilm-l6-v2", "model_name": "Sentence Transformers: all-MiniLM-L6-v2", "context_length": 512, "pricing": { "prompt": "0.000000005", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepcogito/cogito-v2.1-671b@together", "name": "Deep Cogito: Cogito v2.1 671B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepcogito/cogito-v2.1-671b", "canonicalSlug": "deepcogito/cogito-v2.1-671b-20251118", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "structured_outputs", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | deepcogito/cogito-v2.1-671b-20251118", "model_id": "deepcogito/cogito-v2.1-671b", "model_name": "Deep Cogito: Cogito v2.1 671B", "context_length": 128000, "pricing": { "prompt": "0.00000125", "completion": "0.00000125", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.96717005909389, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.1@openai", "name": "OpenAI: GPT-5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.1", "canonicalSlug": "openai/gpt-5.1-20251113", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.1-20251113", "model_id": "openai/gpt-5.1", "model_name": "OpenAI: GPT-5.1", "context_length": 400000, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.000000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/default", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.8799102487122, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.1@openai", "name": "OpenAI: GPT-5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.1", "canonicalSlug": "openai/gpt-5.1-20251113", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.1-20251113", "model_id": "openai/gpt-5.1", "model_name": "OpenAI: GPT-5.1", "context_length": 400000, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000000625", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/default/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.8799102487122, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.1@openai", "name": "OpenAI: GPT-5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.1", "canonicalSlug": "openai/gpt-5.1-20251113", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.1-20251113", "model_id": "openai/gpt-5.1", "model_name": "OpenAI: GPT-5.1", "context_length": 400000, "pricing": { "prompt": "0.0000025", "completion": "0.00002", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/default/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.8799102487122, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.1@azure", "name": "OpenAI: GPT-5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.1", "canonicalSlug": "openai/gpt-5.1-20251113", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.1-20251113", "model_id": "openai/gpt-5.1", "model_name": "OpenAI: GPT-5.1", "context_length": 400000, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.00000013", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.1@azure", "name": "OpenAI: GPT-5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.143, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000143" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.1", "canonicalSlug": "openai/gpt-5.1-20251113", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.1-20251113", "model_id": "openai/gpt-5.1", "model_name": "OpenAI: GPT-5.1", "context_length": 400000, "pricing": { "prompt": "0.000001375", "completion": "0.000011", "web_search": "0.01", "input_cache_read": "0.000000143", "discount": 0 }, "provider_name": "Azure", "tag": "azure/swedencentral", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.1:batch@openai", "name": "OpenAI: GPT-5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.1:batch", "canonicalSlug": "openai/gpt-5.1-20251113", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.1-20251113:batch", "model_id": "openai/gpt-5.1:batch", "model_name": "OpenAI: GPT-5.1", "context_length": 400000, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000000625", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.1:batch@openai", "name": "OpenAI: GPT-5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003125" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.1:batch", "canonicalSlug": "openai/gpt-5.1-20251113", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.1-20251113:batch", "model_id": "openai/gpt-5.1:batch", "model_name": "OpenAI: GPT-5.1", "context_length": 400000, "pricing": { "prompt": "0.0000003125", "completion": "0.0000025", "web_search": "0.01", "input_cache_read": "0.00000003125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.1:batch@openai", "name": "OpenAI: GPT-5.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.1:batch", "canonicalSlug": "openai/gpt-5.1-20251113", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5.1-20251113:batch", "model_id": "openai/gpt-5.1:batch", "model_name": "OpenAI: GPT-5.1", "context_length": 400000, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.000000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/priority", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.1-codex@azure", "name": "OpenAI: GPT-5.1-Codex", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.1-codex", "canonicalSlug": "openai/gpt-5.1-codex-20251113", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.1-codex-20251113", "model_id": "openai/gpt-5.1-codex", "model_name": "OpenAI: GPT-5.1-Codex", "context_length": 400000, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.00000013", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5.1-codex-mini@azure", "name": "OpenAI: GPT-5.1-Codex-Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5.1-codex-mini", "canonicalSlug": "openai/gpt-5.1-codex-mini-20251113", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5.1-codex-mini-20251113", "model_id": "openai/gpt-5.1-codex-mini", "model_name": "OpenAI: GPT-5.1-Codex-Mini", "context_length": 400000, "pricing": { "prompt": "0.00000025", "completion": "0.000002", "web_search": "0.01", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2-thinking@novita", "name": "MoonshotAI: Kimi K2 Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2-thinking", "canonicalSlug": "moonshotai/kimi-k2-thinking-20251106", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 100352, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...", "uptimeLast30m": 99.93174061433447, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | moonshotai/kimi-k2-thinking-20251106", "model_id": "moonshotai/kimi-k2-thinking", "model_name": "MoonshotAI: Kimi K2 Thinking", "context_length": 262144, "pricing": { "prompt": "0.0000006", "completion": "0.0000025", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 100352, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.93174061433447, "uptime_last_5m": 100, "uptime_last_1d": 97.96952224052718, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2-thinking@google", "name": "MoonshotAI: Kimi K2 Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2-thinking", "canonicalSlug": "moonshotai/kimi-k2-thinking-20251106", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | moonshotai/kimi-k2-thinking-20251106", "model_id": "moonshotai/kimi-k2-thinking", "model_name": "MoonshotAI: Kimi K2 Thinking", "context_length": 262144, "pricing": { "prompt": "0.0000006", "completion": "0.0000025", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 98.97725268912009, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "amazon/nova-premier-v1@amazon-bedrock", "name": "Amazon: Nova Premier 1.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "cacheRead": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "amazon/nova-premier-v1", "canonicalSlug": "amazon/nova-premier-v1", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "description": "Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | amazon/nova-premier-v1", "model_id": "amazon/nova-premier-v1", "model_name": "Amazon: Nova Premier 1.0", "context_length": 1000000, "pricing": { "prompt": "0.0000025", "completion": "0.0000125", "input_cache_read": "0.000000625", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-embed-2312@mistral", "name": "Mistral: Mistral Embed 2312", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.0000001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-embed-2312", "canonicalSlug": "mistralai/mistral-embed-2312", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral Embed is a specialized embedding model for text data, optimized for semantic search and RAG applications. Developed by Mistral AI in late 2023, it produces 1024-dimensional vectors that effectively...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/mistral-embed-2312", "model_id": "mistralai/mistral-embed-2312", "model_name": "Mistral: Mistral Embed 2312", "context_length": 8192, "pricing": { "prompt": "0.0000001", "completion": "0", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-embedding-001@google-ai-studio", "name": "Google: Gemini Embedding 001", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-embedding-001", "canonicalSlug": "google/gemini-embedding-001", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 20000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "gemini-embedding-001 provides a unified cutting edge experience across domains, including science, legal, finance, and coding. This embedding model has consistently held a top spot on the Massive Text Embedding Benchmark...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-embedding-001", "model_id": "google/gemini-embedding-001", "model_name": "Google: Gemini Embedding 001", "context_length": 20000, "pricing": { "prompt": "0.00000015", "completion": "0", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.92386069958445, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-embedding-001@google", "name": "Google: Gemini Embedding 001", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-embedding-001", "canonicalSlug": "google/gemini-embedding-001", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 20000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "gemini-embedding-001 provides a unified cutting edge experience across domains, including science, legal, finance, and coding. This embedding model has consistently held a top spot on the Massive Text Embedding Benchmark...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-embedding-001", "model_id": "google/gemini-embedding-001", "model_name": "Google: Gemini Embedding 001", "context_length": 20000, "pricing": { "prompt": "0.00000015", "completion": "0", "image": "0", "audio": "0", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us-central1", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99901722584697, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/text-embedding-ada-002@openai", "name": "OpenAI: Text Embedding Ada 002", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.0000001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/text-embedding-ada-002", "canonicalSlug": "openai/text-embedding-ada-002", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "text-embedding-ada-002 is OpenAI's legacy text embedding model.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/text-embedding-ada-002", "model_id": "openai/text-embedding-ada-002", "model_name": "OpenAI: Text Embedding Ada 002", "context_length": 8192, "pricing": { "prompt": "0.0000001", "completion": "0", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/text-embedding-ada-002:batch@openai", "name": "OpenAI: Text Embedding Ada 002", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000005" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/text-embedding-ada-002:batch", "canonicalSlug": "openai/text-embedding-ada-002", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "text-embedding-ada-002 is OpenAI's legacy text embedding model.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/text-embedding-ada-002:batch", "model_id": "openai/text-embedding-ada-002:batch", "model_name": "OpenAI: Text Embedding Ada 002", "context_length": 8192, "pricing": { "prompt": "0.00000005", "completion": "0", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/codestral-embed-2505@mistral", "name": "Mistral: Codestral Embed 2505", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000015" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/codestral-embed-2505", "canonicalSlug": "mistralai/codestral-embed-2505", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral Codestral Embed is specially designed for code, perfect for embedding code databases, repositories, and powering coding assistants with state-of-the-art retrieval.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/codestral-embed-2505", "model_id": "mistralai/codestral-embed-2505", "model_name": "Mistral: Codestral Embed 2505", "context_length": 8192, "pricing": { "prompt": "0.00000015", "completion": "0", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/text-embedding-3-large@openai", "name": "OpenAI: Text Embedding 3 Large", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000013" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/text-embedding-3-large", "canonicalSlug": "openai/text-embedding-3-large", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "text-embedding-3-large is OpenAI's most capable embedding model for both english and non-english tasks. Embeddings are a numerical representation of text that can be used to measure the relatedness between two...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/text-embedding-3-large", "model_id": "openai/text-embedding-3-large", "model_name": "OpenAI: Text Embedding 3 Large", "context_length": 8192, "pricing": { "prompt": "0.00000013", "completion": "0", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.98919786371869, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/text-embedding-3-large@azure", "name": "OpenAI: Text Embedding 3 Large", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000013" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/text-embedding-3-large", "canonicalSlug": "openai/text-embedding-3-large", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "max_completion_tokens" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "text-embedding-3-large is OpenAI's most capable embedding model for both english and non-english tasks. Embeddings are a numerical representation of text that can be used to measure the relatedness between two...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/text-embedding-3-large", "model_id": "openai/text-embedding-3-large", "model_name": "OpenAI: Text Embedding 3 Large", "context_length": 8192, "pricing": { "prompt": "0.00000013", "completion": "0", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "max_completion_tokens" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/text-embedding-3-large:batch@openai", "name": "OpenAI: Text Embedding 3 Large", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.065, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.000000065" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/text-embedding-3-large:batch", "canonicalSlug": "openai/text-embedding-3-large", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "text-embedding-3-large is OpenAI's most capable embedding model for both english and non-english tasks. Embeddings are a numerical representation of text that can be used to measure the relatedness between two...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/text-embedding-3-large:batch", "model_id": "openai/text-embedding-3-large:batch", "model_name": "OpenAI: Text Embedding 3 Large", "context_length": 8192, "pricing": { "prompt": "0.000000065", "completion": "0", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/text-embedding-3-small@openai", "name": "OpenAI: Text Embedding 3 Small", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000002" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/text-embedding-3-small", "canonicalSlug": "openai/text-embedding-3-small", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "text-embedding-3-small is OpenAI's improved, more performant version of the ada embedding model. Embeddings are a numerical representation of text that can be used to measure the relatedness between two pieces...", "uptimeLast30m": 99.9992042366441, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/text-embedding-3-small", "model_id": "openai/text-embedding-3-small", "model_name": "OpenAI: Text Embedding 3 Small", "context_length": 8192, "pricing": { "prompt": "0.00000002", "completion": "0", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.9992042366441, "uptime_last_5m": 100, "uptime_last_1d": 99.99960270222084, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/text-embedding-3-small@azure", "name": "OpenAI: Text Embedding 3 Small", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000002" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/text-embedding-3-small", "canonicalSlug": "openai/text-embedding-3-small", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "max_completion_tokens" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "text-embedding-3-small is OpenAI's improved, more performant version of the ada embedding model. Embeddings are a numerical representation of text that can be used to measure the relatedness between two pieces...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/text-embedding-3-small", "model_id": "openai/text-embedding-3-small", "model_name": "OpenAI: Text Embedding 3 Small", "context_length": 8192, "pricing": { "prompt": "0.00000002", "completion": "0", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "max_completion_tokens" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "perplexity/sonar-pro-search@perplexity", "name": "Perplexity: Sonar Pro Search", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.018, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.018" } ] } ], "metadata": { "source": "openrouter", "modelId": "perplexity/sonar-pro-search", "canonicalSlug": "perplexity/sonar-pro-search", "servingProvider": "Perplexity", "servingProviderSlug": "perplexity", "contextLength": 200000, "maxCompletionTokens": 8000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options", "structured_outputs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Perplexity | perplexity/sonar-pro-search", "model_id": "perplexity/sonar-pro-search", "model_name": "Perplexity: Sonar Pro Search", "context_length": 200000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.018", "discount": 0 }, "provider_name": "Perplexity", "tag": "perplexity", "quantization": "unknown", "max_completion_tokens": 8000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.98482319016543, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/voxtral-small-24b-2507@mistral", "name": "Mistral: Voxtral Small 24B 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/voxtral-small-24b-2507", "canonicalSlug": "mistralai/voxtral-small-24b-2507", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+file+audio->text", "input_modalities": [ "text", "audio", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...", "uptimeLast30m": 98.85057471264368, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/voxtral-small-24b-2507", "model_id": "mistralai/voxtral-small-24b-2507", "model_name": "Mistral: Voxtral Small 24B 2507", "context_length": 32000, "pricing": { "prompt": "0.0000001", "completion": "0.0000003", "audio": "0.0001", "input_cache_read": "0.00000001", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 98.85057471264368, "uptime_last_5m": 100, "uptime_last_1d": 99.44636678200692, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-safeguard-20b@groq", "name": "OpenAI: gpt-oss-safeguard-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-safeguard-20b", "canonicalSlug": "openai/gpt-oss-safeguard-20b", "servingProvider": "Groq", "servingProviderSlug": "groq", "contextLength": 131072, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Groq | openai/gpt-oss-safeguard-20b", "model_id": "openai/gpt-oss-safeguard-20b", "model_name": "OpenAI: gpt-oss-safeguard-20b", "context_length": 131072, "pricing": { "prompt": "0.000000075", "completion": "0.0000003", "input_cache_read": "0.0000000375", "discount": 0 }, "provider_name": "Groq", "tag": "groq", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-embedding-8b@nebius", "name": "Qwen: Qwen3 Embedding 8B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-embedding-8b", "canonicalSlug": "qwen/qwen3-embedding-8b", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 32000, "maxCompletionTokens": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | qwen/qwen3-embedding-8b", "model_id": "qwen/qwen3-embedding-8b", "model_name": "Qwen: Qwen3 Embedding 8B", "context_length": 32000, "pricing": { "prompt": "0.00000001", "completion": "0", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius", "quantization": "unknown", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99748247699088, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-embedding-8b@deepinfra", "name": "Qwen: Qwen3 Embedding 8B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-embedding-8b", "canonicalSlug": "qwen/qwen3-embedding-8b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...", "uptimeLast30m": 99.9965528731802, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-embedding-8b", "model_id": "qwen/qwen3-embedding-8b", "model_name": "Qwen: Qwen3 Embedding 8B", "context_length": 32768, "pricing": { "prompt": "0.00000001", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 99.9965528731802, "uptime_last_5m": 100, "uptime_last_1d": 99.99720154121435, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-embedding-8b@siliconflow", "name": "Qwen: Qwen3 Embedding 8B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000004" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-embedding-8b", "canonicalSlug": "qwen/qwen3-embedding-8b", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "temperature", "top_p", "top_k", "frequency_penalty" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...", "uptimeLast30m": 99.67815735391058, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3-embedding-8b", "model_id": "qwen/qwen3-embedding-8b", "model_name": "Qwen: Qwen3 Embedding 8B", "context_length": 32768, "pricing": { "prompt": "0.00000004", "completion": "0", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "temperature", "top_p", "top_k", "frequency_penalty" ], "status": 0, "uptime_last_30m": 99.67815735391058, "uptime_last_5m": 100, "uptime_last_1d": 99.96386172889542, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-nano-12b-v2-vl:free@nvidia", "name": "NVIDIA: Nemotron Nano 12B 2 VL", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-nano-12b-v2-vl:free", "canonicalSlug": "nvidia/nemotron-nano-12b-v2-vl", "servingProvider": "Nvidia", "servingProviderSlug": "nvidia", "contextLength": 128000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tool_choice", "tools" ], "architecture": { "modality": "text+image+video->text", "input_modalities": [ "image", "text", "video" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA Nemotron Nano 2 VL is a 12-billion-parameter open multimodal reasoning model designed for video understanding and document intelligence. It introduces a hybrid Transformer-Mamba architecture, combining transformer-level accuracy with Mamba’s...", "uptimeLast30m": 81.10321051497522, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nvidia | nvidia/nemotron-nano-12b-v2-vl:free", "model_id": "nvidia/nemotron-nano-12b-v2-vl:free", "model_name": "NVIDIA: Nemotron Nano 12B 2 VL", "context_length": 128000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Nvidia", "tag": "nvidia", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "max_tokens", "seed", "top_p", "tool_choice", "tools" ], "status": -2, "uptime_last_30m": 81.10321051497522, "uptime_last_5m": 85.71428571428571, "uptime_last_1d": 76.81710246192478, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-embedding-4b@deepinfra", "name": "Qwen: Qwen3 Embedding 4B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00000002" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-embedding-4b", "canonicalSlug": "qwen/qwen3-embedding-4b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "architecture": { "modality": "text->embeddings", "input_modalities": [ "text" ], "output_modalities": [ "embeddings" ], "tokenizer": "Other", "instruct_type": null }, "description": "The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-embedding-4b", "model_id": "qwen/qwen3-embedding-4b", "model_name": "Qwen: Qwen3 Embedding 4B", "context_length": 32768, "pricing": { "prompt": "0.00000002", "completion": "0", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2@minimax", "name": "MiniMax: MiniMax M2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.255, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000255" } ], "output": [ { "amount": 1.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000102" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2", "canonicalSlug": "minimax/minimax-m2", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tool_choice", "tools" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/minimax-m2", "model_id": "minimax/minimax-m2", "model_name": "MiniMax: MiniMax M2", "context_length": 204800, "pricing": { "prompt": "0.000000255", "completion": "0.00000102", "discount": 0.15 }, "provider_name": "Minimax", "tag": "minimax/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.96742671009771, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2@google", "name": "MiniMax: MiniMax M2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2", "canonicalSlug": "minimax/minimax-m2", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 196608, "maxCompletionTokens": 196608, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | minimax/minimax-m2", "model_id": "minimax/minimax-m2", "model_name": "MiniMax: MiniMax M2", "context_length": 196608, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": 196608, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.98496014438261, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m2@novita", "name": "MiniMax: MiniMax M2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m2", "canonicalSlug": "minimax/minimax-m2", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | minimax/minimax-m2", "model_id": "minimax/minimax-m2", "model_name": "MiniMax: MiniMax M2", "context_length": 204800, "pricing": { "prompt": "0.0000003", "completion": "0.0000012", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.57698815566836, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-32b-instruct@alibaba", "name": "Qwen: Qwen3 VL 32B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.10400000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000104" } ], "output": [ { "amount": 0.41600000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000416" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-32b-instruct", "canonicalSlug": "qwen/qwen3-vl-32b-instruct", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-vl-32b-instruct", "model_id": "qwen/qwen3-vl-32b-instruct", "model_name": "Qwen: Qwen3 VL 32B Instruct", "context_length": 131072, "pricing": { "prompt": "0.000000104", "completion": "0.000000416", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 129024, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9999139105117, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "ibm-granite/granite-4.0-h-micro@cloudflare", "name": "IBM: Granite 4.0 Micro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.017, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000017" } ], "output": [ { "amount": 0.112, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000112" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "ibm-granite/granite-4.0-h-micro", "canonicalSlug": "ibm-granite/granite-4.0-h-micro", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 131000, "maxCompletionTokens": 131000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | ibm-granite/granite-4.0-h-micro", "model_id": "ibm-granite/granite-4.0-h-micro", "model_name": "IBM: Granite 4.0 Micro", "context_length": 131000, "pricing": { "prompt": "0.000000017", "completion": "0.000000112", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 131000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-image-mini@openai", "name": "OpenAI: GPT-5 Image Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-image-mini", "canonicalSlug": "openai/gpt-5-image-mini", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+file->text+image", "input_modalities": [ "file", "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-image-mini", "model_id": "openai/gpt-5-image-mini", "model_name": "OpenAI: GPT-5 Image Mini", "context_length": 400000, "pricing": { "prompt": "0.0000025", "completion": "0.000002", "image_output": "0.000008", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-haiku-4.5@google", "name": "Anthropic: Claude Haiku 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-haiku-4.5", "canonicalSlug": "anthropic/claude-4.5-haiku-20251001", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...", "uptimeLast30m": 99.93098688750862, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.5-haiku-20251001", "model_id": "anthropic/claude-haiku-4.5", "model_name": "Anthropic: Claude Haiku 4.5", "context_length": 200000, "pricing": { "prompt": "0.000001", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000001", "input_cache_write": "0.00000125", "input_cache_write_1h": "0.000002", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k" ], "status": 0, "uptime_last_30m": 99.93098688750862, "uptime_last_5m": 99.43342776203966, "uptime_last_1d": 99.90769855579006, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-haiku-4.5@anthropic", "name": "Anthropic: Claude Haiku 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-haiku-4.5", "canonicalSlug": "anthropic/claude-4.5-haiku-20251001", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...", "uptimeLast30m": 99.51523545706371, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.5-haiku-20251001", "model_id": "anthropic/claude-haiku-4.5", "model_name": "Anthropic: Claude Haiku 4.5", "context_length": 200000, "pricing": { "prompt": "0.000001", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000001", "input_cache_write": "0.00000125", "input_cache_write_1h": "0.000002", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 99.51523545706371, "uptime_last_5m": 99.625468164794, "uptime_last_1d": 99.4125612974492, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-haiku-4.5@azure", "name": "Anthropic: Claude Haiku 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-haiku-4.5", "canonicalSlug": "anthropic/claude-4.5-haiku-20251001", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "top_k", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-4.5-haiku-20251001", "model_id": "anthropic/claude-haiku-4.5", "model_name": "Anthropic: Claude Haiku 4.5", "context_length": 200000, "pricing": { "prompt": "0.000001", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000001", "input_cache_write": "0.00000125", "input_cache_write_1h": "0.000002", "discount": 0 }, "provider_name": "Azure", "tag": "azure/global", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "top_k", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.8731858531774, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-haiku-4.5@amazon-bedrock", "name": "Anthropic: Claude Haiku 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-haiku-4.5", "canonicalSlug": "anthropic/claude-4.5-haiku-20251001", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...", "uptimeLast30m": 99.91259527757425, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.5-haiku-20251001", "model_id": "anthropic/claude-haiku-4.5", "model_name": "Anthropic: Claude Haiku 4.5", "context_length": 200000, "pricing": { "prompt": "0.000001", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000001", "input_cache_write": "0.00000125", "input_cache_write_1h": "0.000002", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/global", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 99.91259527757425, "uptime_last_5m": 99.9298081422555, "uptime_last_1d": 99.93482708507548, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-haiku-4.5@google", "name": "Anthropic: Claude Haiku 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-haiku-4.5", "canonicalSlug": "anthropic/claude-4.5-haiku-20251001", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.5-haiku-20251001", "model_id": "anthropic/claude-haiku-4.5", "model_name": "Anthropic: Claude Haiku 4.5", "context_length": 200000, "pricing": { "prompt": "0.0000011", "completion": "0.0000055", "web_search": "0.01", "input_cache_read": "0.00000011", "input_cache_write": "0.000001375", "input_cache_write_1h": "0.0000022", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/europe", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-haiku-4.5@google", "name": "Anthropic: Claude Haiku 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-haiku-4.5", "canonicalSlug": "anthropic/claude-4.5-haiku-20251001", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.5-haiku-20251001", "model_id": "anthropic/claude-haiku-4.5", "model_name": "Anthropic: Claude Haiku 4.5", "context_length": 200000, "pricing": { "prompt": "0.0000011", "completion": "0.0000055", "web_search": "0.01", "input_cache_read": "0.00000011", "input_cache_write": "0.000001375", "input_cache_write_1h": "0.0000022", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us-east5", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-haiku-4.5@amazon-bedrock", "name": "Anthropic: Claude Haiku 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-haiku-4.5", "canonicalSlug": "anthropic/claude-4.5-haiku-20251001", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.5-haiku-20251001", "model_id": "anthropic/claude-haiku-4.5", "model_name": "Anthropic: Claude Haiku 4.5", "context_length": 200000, "pricing": { "prompt": "0.0000011", "completion": "0.0000055", "web_search": "0.01", "input_cache_read": "0.00000011", "input_cache_write": "0.000001375", "input_cache_write_1h": "0.0000022", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-haiku-4.5@amazon-bedrock", "name": "Anthropic: Claude Haiku 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-haiku-4.5", "canonicalSlug": "anthropic/claude-4.5-haiku-20251001", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.5-haiku-20251001", "model_id": "anthropic/claude-haiku-4.5", "model_name": "Anthropic: Claude Haiku 4.5", "context_length": 200000, "pricing": { "prompt": "0.0000011", "completion": "0.0000055", "web_search": "0.01", "input_cache_read": "0.00000011", "input_cache_write": "0.000001375", "input_cache_write_1h": "0.0000022", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/us", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-haiku-4.5:batch@anthropic", "name": "Anthropic: Claude Haiku 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-haiku-4.5:batch", "canonicalSlug": "anthropic/claude-4.5-haiku-20251001", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.5-haiku-20251001:batch", "model_id": "anthropic/claude-haiku-4.5:batch", "model_name": "Anthropic: Claude Haiku 4.5", "context_length": 200000, "pricing": { "prompt": "0.0000005", "completion": "0.0000025", "web_search": "0.01", "input_cache_read": "0.00000005", "input_cache_write": "0.000000625", "input_cache_write_1h": "0.000001", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-8b-thinking@alibaba", "name": "Qwen: Qwen3 VL 8B Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "output": [ { "amount": 2.0999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000021" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-8b-thinking", "canonicalSlug": "qwen/qwen3-vl-8b-thinking", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-vl-8b-thinking", "model_id": "qwen/qwen3-vl-8b-thinking", "model_name": "Qwen: Qwen3 VL 8B Thinking", "context_length": 131072, "pricing": { "prompt": "0.00000018", "completion": "0.0000021", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 126976, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-8b-instruct@alibaba", "name": "Qwen: Qwen3 VL 8B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.117, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000117" } ], "output": [ { "amount": 0.45499999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000455" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-8b-instruct", "canonicalSlug": "qwen/qwen3-vl-8b-instruct", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-vl-8b-instruct", "model_id": "qwen/qwen3-vl-8b-instruct", "model_name": "Qwen: Qwen3 VL 8B Instruct", "context_length": 131072, "pricing": { "prompt": "0.000000117", "completion": "0.000000455", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 129024, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9827918857965, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-8b-instruct@parasail", "name": "Qwen: Qwen3 VL 8B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-8b-instruct", "canonicalSlug": "qwen/qwen3-vl-8b-instruct", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...", "uptimeLast30m": 97.1401028277635, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | qwen/qwen3-vl-8b-instruct", "model_id": "qwen/qwen3-vl-8b-instruct", "model_name": "Qwen: Qwen3 VL 8B Instruct", "context_length": 262144, "pricing": { "prompt": "0.00000025", "completion": "0.00000075", "input_cache_read": "0.00000012", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 97.1401028277635, "uptime_last_5m": 89.60000000000001, "uptime_last_1d": 99.43574179069802, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-image@openai", "name": "OpenAI: GPT-5 Image", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-image", "canonicalSlug": "openai/gpt-5-image", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image+file->text+image", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "image", "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-image", "model_id": "openai/gpt-5-image", "model_name": "OpenAI: GPT-5 Image", "context_length": 400000, "pricing": { "prompt": "0.00001", "completion": "0.00001", "image_output": "0.00004", "web_search": "0.01", "input_cache_read": "0.00000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash-image@google-ai-studio", "name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000003" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000003" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 3e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000003" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash-image", "canonicalSlug": "google/gemini-2.5-flash-image", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 32768, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-flash-image", "model_id": "google/gemini-2.5-flash-image", "model_name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "context_length": 32768, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "image": "0.0000003", "image_output": "0.00003", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.0000025", "input_cache_read": "0.00000003", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash-image@google-ai-studio", "name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000015" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000000015" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 1.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000015" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash-image", "canonicalSlug": "google/gemini-2.5-flash-image", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 32768, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-flash-image", "model_id": "google/gemini-2.5-flash-image", "model_name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "context_length": 32768, "pricing": { "prompt": "0.00000015", "completion": "0.00000125", "image": "0.00000015", "image_output": "0.000015", "audio": "0.0000005", "input_audio_cache": "0.00000005", "web_search": "0.014", "internal_reasoning": "0.00000125", "input_cache_read": "0.000000015", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash-image@google-ai-studio", "name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000054" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.054, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000000054" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000015" } ], "other": [ { "amount": 5.4e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000054" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash-image", "canonicalSlug": "google/gemini-2.5-flash-image", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 32768, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-flash-image", "model_id": "google/gemini-2.5-flash-image", "model_name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "context_length": 32768, "pricing": { "prompt": "0.00000054", "completion": "0.0000045", "image": "0.00000054", "image_output": "0.000054", "audio": "0.0000018", "input_audio_cache": "0.00000018", "web_search": "0.014", "internal_reasoning": "0.0000045", "input_cache_read": "0.000000054", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash-image@google", "name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000003" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000003" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 3e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000003" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash-image", "canonicalSlug": "google/gemini-2.5-flash-image", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 32768, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...", "uptimeLast30m": 99.73890339425587, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-flash-image", "model_id": "google/gemini-2.5-flash-image", "model_name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "context_length": 32768, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "image": "0.0000003", "image_output": "0.00003", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.0000025", "input_cache_read": "0.00000003", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.73890339425587, "uptime_last_5m": 100, "uptime_last_1d": 99.71509205751914, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash-image@google", "name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000015" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000000015" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 1.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000015" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash-image", "canonicalSlug": "google/gemini-2.5-flash-image", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 32768, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs" ], "architecture": { "modality": "text+image->text+image", "input_modalities": [ "image", "text" ], "output_modalities": [ "image", "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...", "uptimeLast30m": 99.73890339425587, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-flash-image", "model_id": "google/gemini-2.5-flash-image", "model_name": "Google: Nano Banana (Gemini 2.5 Flash Image)", "context_length": 32768, "pricing": { "prompt": "0.00000015", "completion": "0.00000125", "image": "0.00000015", "image_output": "0.000015", "audio": "0.0000005", "input_audio_cache": "0.00000005", "web_search": "0.014", "internal_reasoning": "0.00000125", "input_cache_read": "0.000000015", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/flex", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format", "stop", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.73890339425587, "uptime_last_5m": 100, "uptime_last_1d": 99.71509205751914, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-30b-a3b-thinking@alibaba", "name": "Qwen: Qwen3 VL 30B A3B Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-30b-a3b-thinking", "canonicalSlug": "qwen/qwen3-vl-30b-a3b-thinking", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-vl-30b-a3b-thinking", "model_id": "qwen/qwen3-vl-30b-a3b-thinking", "model_name": "Qwen: Qwen3 VL 30B A3B Thinking", "context_length": 131072, "pricing": { "prompt": "0.0000002", "completion": "0.0000024", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 126976, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-30b-a3b-thinking@siliconflow", "name": "Qwen: Qwen3 VL 30B A3B Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000029" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-30b-a3b-thinking", "canonicalSlug": "qwen/qwen3-vl-30b-a3b-thinking", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...", "uptimeLast30m": 99.58904109589041, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3-vl-30b-a3b-thinking", "model_id": "qwen/qwen3-vl-30b-a3b-thinking", "model_name": "Qwen: Qwen3 VL 30B A3B Thinking", "context_length": 262144, "pricing": { "prompt": "0.00000029", "completion": "0.000001", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 99.58904109589041, "uptime_last_5m": 100, "uptime_last_1d": 99.8690119760479, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-30b-a3b-instruct@alibaba", "name": "Qwen: Qwen3 VL 30B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000052" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-30b-a3b-instruct", "canonicalSlug": "qwen/qwen3-vl-30b-a3b-instruct", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-vl-30b-a3b-instruct", "model_id": "qwen/qwen3-vl-30b-a3b-instruct", "model_name": "Qwen: Qwen3 VL 30B A3B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000013", "completion": "0.00000052", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 129024, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.97128625842367, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-30b-a3b-instruct@deepinfra", "name": "Qwen: Qwen3 VL 30B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-30b-a3b-instruct", "canonicalSlug": "qwen/qwen3-vl-30b-a3b-instruct", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...", "uptimeLast30m": 99.907019990702, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-vl-30b-a3b-instruct", "model_id": "qwen/qwen3-vl-30b-a3b-instruct", "model_name": "Qwen: Qwen3 VL 30B A3B Instruct", "context_length": 262144, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.907019990702, "uptime_last_5m": 100, "uptime_last_1d": 99.77586089791482, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-30b-a3b-instruct@novita", "name": "Qwen: Qwen3 VL 30B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-30b-a3b-instruct", "canonicalSlug": "qwen/qwen3-vl-30b-a3b-instruct", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "bf16", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...", "uptimeLast30m": 97.67441860465115, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3-vl-30b-a3b-instruct", "model_id": "qwen/qwen3-vl-30b-a3b-instruct", "model_name": "Qwen: Qwen3 VL 30B A3B Instruct", "context_length": 131072, "pricing": { "prompt": "0.0000002", "completion": "0.0000007", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 97.67441860465115, "uptime_last_5m": 97.31182795698925, "uptime_last_1d": 96.0784973744446, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-30b-a3b-instruct@siliconflow", "name": "Qwen: Qwen3 VL 30B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000029" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-30b-a3b-instruct", "canonicalSlug": "qwen/qwen3-vl-30b-a3b-instruct", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...", "uptimeLast30m": 99.09502262443439, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3-vl-30b-a3b-instruct", "model_id": "qwen/qwen3-vl-30b-a3b-instruct", "model_name": "Qwen: Qwen3 VL 30B A3B Instruct", "context_length": 262144, "pricing": { "prompt": "0.00000029", "completion": "0.000001", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 99.09502262443439, "uptime_last_5m": 100, "uptime_last_1d": 96.1803576055973, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-pro@openai", "name": "OpenAI: GPT-5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-pro", "canonicalSlug": "openai/gpt-5-pro-2025-10-06", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-pro-2025-10-06", "model_id": "openai/gpt-5-pro", "model_name": "OpenAI: GPT-5 Pro", "context_length": 400000, "pricing": { "prompt": "0.000015", "completion": "0.00012", "web_search": "0.01", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-pro:batch@openai", "name": "OpenAI: GPT-5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-pro:batch", "canonicalSlug": "openai/gpt-5-pro-2025-10-06", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-pro-2025-10-06:batch", "model_id": "openai/gpt-5-pro:batch", "model_name": "OpenAI: GPT-5 Pro", "context_length": 400000, "pricing": { "prompt": "0.0000075", "completion": "0.00006", "web_search": "0.01", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.6@venice", "name": "Z.ai: GLM 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.43, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000043" } ], "output": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.6", "canonicalSlug": "z-ai/glm-4.6", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 198000, "maxCompletionTokens": 16384, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...", "uptimeLast30m": 99.82300884955752, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | z-ai/glm-4.6", "model_id": "z-ai/glm-4.6", "model_name": "Z.ai: GLM 4.6", "context_length": 198000, "pricing": { "prompt": "0.00000043", "completion": "0.00000175", "input_cache_read": "0.00000008", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp4", "quantization": "fp4", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.82300884955752, "uptime_last_5m": 100, "uptime_last_1d": 99.16531212933269, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.6@deepinfra", "name": "Z.ai: GLM 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.6", "canonicalSlug": "z-ai/glm-4.6", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | z-ai/glm-4.6", "model_id": "z-ai/glm-4.6", "model_name": "Z.ai: GLM 4.6", "context_length": 202752, "pricing": { "prompt": "0.0000005", "completion": "0.000002", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.91095761042605, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.6@novita", "name": "Z.ai: GLM 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.6", "canonicalSlug": "z-ai/glm-4.6", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 204800, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...", "uptimeLast30m": 96.1352657004831, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | z-ai/glm-4.6", "model_id": "z-ai/glm-4.6", "model_name": "Z.ai: GLM 4.6", "context_length": 204800, "pricing": { "prompt": "0.00000055", "completion": "0.0000022", "input_cache_read": "0.00000011", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": 96.1352657004831, "uptime_last_5m": 95.1219512195122, "uptime_last_1d": 97.16146054983403, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.6@z.ai", "name": "Z.ai: GLM 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.6", "canonicalSlug": "z-ai/glm-4.6", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 202752, "maxCompletionTokens": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...", "uptimeLast30m": 97.75784753363229, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-4.6", "model_id": "z-ai/glm-4.6", "model_name": "Z.ai: GLM 4.6", "context_length": 202752, "pricing": { "prompt": "0.0000006", "completion": "0.0000022", "input_cache_read": "0.00000011", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai/fp4", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k" ], "status": 0, "uptime_last_30m": 97.75784753363229, "uptime_last_5m": null, "uptime_last_1d": 90.48667694340453, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.6@atlascloud", "name": "Z.ai: GLM 4.6", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.6", "canonicalSlug": "z-ai/glm-4.6", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 202752, "maxCompletionTokens": 202752, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...", "uptimeLast30m": 99.43820224719101, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | z-ai/glm-4.6", "model_id": "z-ai/glm-4.6", "model_name": "Z.ai: GLM 4.6", "context_length": 202752, "pricing": { "prompt": "0.0000006", "completion": "0.0000022", "input_cache_read": "0.00000011", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 202752, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.43820224719101, "uptime_last_5m": null, "uptime_last_1d": 99.53190928270043, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.5@amazon-bedrock", "name": "Anthropic: Claude Sonnet 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.5", "canonicalSlug": "anthropic/claude-4.5-sonnet-20250929", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...", "uptimeLast30m": 99.90612091625985, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.5-sonnet-20250929", "model_id": "anthropic/claude-sonnet-4.5", "model_name": "Anthropic: Claude Sonnet 4.5", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "input_cache_write": "0.0000075", "input_cache_write_1h": "0.000012" } ] }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 99.90612091625985, "uptime_last_5m": 100, "uptime_last_1d": 99.90328402545703, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.5@anthropic", "name": "Anthropic: Claude Sonnet 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.5", "canonicalSlug": "anthropic/claude-4.5-sonnet-20250929", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.5-sonnet-20250929", "model_id": "anthropic/claude-sonnet-4.5", "model_name": "Anthropic: Claude Sonnet 4.5", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "input_cache_write": "0.0000075", "input_cache_write_1h": "0.000012" } ] }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.37210563644707, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.5@google", "name": "Anthropic: Claude Sonnet 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.5", "canonicalSlug": "anthropic/claude-4.5-sonnet-20250929", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...", "uptimeLast30m": 99.87012987012987, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.5-sonnet-20250929", "model_id": "anthropic/claude-sonnet-4.5", "model_name": "Anthropic: Claude Sonnet 4.5", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "input_cache_write": "0.0000075", "input_cache_write_1h": "0.000012" } ] }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k" ], "status": 0, "uptime_last_30m": 99.87012987012987, "uptime_last_5m": 100, "uptime_last_1d": 99.98038104372847, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.5@claude-platform-on-aws", "name": "Anthropic: Claude Sonnet 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.5", "canonicalSlug": "anthropic/claude-4.5-sonnet-20250929", "servingProvider": "Claude Platform on AWS", "servingProviderSlug": "claude-platform-on-aws", "contextLength": 1000000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "top_k", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Claude Platform on AWS | anthropic/claude-4.5-sonnet-20250929", "model_id": "anthropic/claude-sonnet-4.5", "model_name": "Anthropic: Claude Sonnet 4.5", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "input_cache_write": "0.0000075", "input_cache_write_1h": "0.000012" } ] }, "provider_name": "Claude Platform on AWS", "tag": "claude-on-aws", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tool_choice", "tools", "structured_outputs", "top_k", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.63540907102232, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.5@azure", "name": "Anthropic: Claude Sonnet 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.5", "canonicalSlug": "anthropic/claude-4.5-sonnet-20250929", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "top_k", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | anthropic/claude-4.5-sonnet-20250929", "model_id": "anthropic/claude-sonnet-4.5", "model_name": "Anthropic: Claude Sonnet 4.5", "context_length": 200000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0 }, "provider_name": "Azure", "tag": "azure/global", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "max_completion_tokens", "tools", "tool_choice", "top_k", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.5@google", "name": "Anthropic: Claude Sonnet 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000033" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.5", "canonicalSlug": "anthropic/claude-4.5-sonnet-20250929", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.5-sonnet-20250929", "model_id": "anthropic/claude-sonnet-4.5", "model_name": "Anthropic: Claude Sonnet 4.5", "context_length": 1000000, "pricing": { "prompt": "0.0000033", "completion": "0.0000165", "web_search": "0.01", "input_cache_read": "0.00000033", "input_cache_write": "0.000004125", "input_cache_write_1h": "0.0000066", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000066", "completion": "0.00002475", "input_cache_read": "0.00000066", "input_cache_write": "0.00000825", "input_cache_write_1h": "0.0000132" } ] }, "provider_name": "Google", "tag": "google-vertex/us-east5", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.5@amazon-bedrock", "name": "Anthropic: Claude Sonnet 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000033" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.5", "canonicalSlug": "anthropic/claude-4.5-sonnet-20250929", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 1000000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.5-sonnet-20250929", "model_id": "anthropic/claude-sonnet-4.5", "model_name": "Anthropic: Claude Sonnet 4.5", "context_length": 1000000, "pricing": { "prompt": "0.0000033", "completion": "0.0000165", "web_search": "0.01", "input_cache_read": "0.00000033", "input_cache_write": "0.000004125", "input_cache_write_1h": "0.0000066", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000066", "completion": "0.00002475", "input_cache_read": "0.00000066", "input_cache_write": "0.00000825", "input_cache_write_1h": "0.0000132" } ] }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4.5:batch@anthropic", "name": "Anthropic: Claude Sonnet 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4.5:batch", "canonicalSlug": "anthropic/claude-4.5-sonnet-20250929", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 1000000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.5-sonnet-20250929:batch", "model_id": "anthropic/claude-sonnet-4.5:batch", "model_name": "Anthropic: Claude Sonnet 4.5", "context_length": 1000000, "pricing": { "prompt": "0.0000015", "completion": "0.0000075", "web_search": "0.01", "input_cache_read": "0.00000015", "input_cache_write": "0.000001875", "input_cache_write_1h": "0.000003", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000003", "completion": "0.00001125", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006" } ] }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "top_k", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2-exp@novita", "name": "DeepSeek: DeepSeek V3.2 Exp", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 0.41, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000041" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2-exp", "canonicalSlug": "deepseek/deepseek-v3.2-exp", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 163840, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 99.89177489177489, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-v3.2-exp", "model_id": "deepseek/deepseek-v3.2-exp", "model_name": "DeepSeek: DeepSeek V3.2 Exp", "context_length": 163840, "pricing": { "prompt": "0.00000027", "completion": "0.00000041", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.89177489177489, "uptime_last_5m": 100, "uptime_last_1d": 99.85112074351765, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2-exp@siliconflow", "name": "DeepSeek: DeepSeek V3.2 Exp", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 0.41, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000041" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2-exp", "canonicalSlug": "deepseek/deepseek-v3.2-exp", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 163840, "maxCompletionTokens": 163840, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | deepseek/deepseek-v3.2-exp", "model_id": "deepseek/deepseek-v3.2-exp", "model_name": "DeepSeek: DeepSeek V3.2 Exp", "context_length": 163840, "pricing": { "prompt": "0.00000027", "completion": "0.00000041", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 163840, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.88083889418495, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.2-exp@atlascloud", "name": "DeepSeek: DeepSeek V3.2 Exp", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 0.41, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000041" } ], "cacheRead": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.2-exp", "canonicalSlug": "deepseek/deepseek-v3.2-exp", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 163840, "maxCompletionTokens": 163840, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...", "uptimeLast30m": 99.89539748953975, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | deepseek/deepseek-v3.2-exp", "model_id": "deepseek/deepseek-v3.2-exp", "model_name": "DeepSeek: DeepSeek V3.2 Exp", "context_length": 163840, "pricing": { "prompt": "0.00000027", "completion": "0.00000041", "input_cache_read": "0.00000027", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 163840, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.89539748953975, "uptime_last_5m": 100, "uptime_last_1d": 99.30471584038693, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thedrummer/cydonia-24b-v4.1@parasail", "name": "TheDrummer: Cydonia 24B V4.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thedrummer/cydonia-24b-v4.1", "canonicalSlug": "thedrummer/cydonia-24b-v4.1", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | thedrummer/cydonia-24b-v4.1", "model_id": "thedrummer/cydonia-24b-v4.1", "model_name": "TheDrummer: Cydonia 24B V4.1", "context_length": 131072, "pricing": { "prompt": "0.0000003", "completion": "0.0000005", "input_cache_read": "0.00000015", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.84368149150733, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "relace/relace-apply-3@relace", "name": "Relace: Relace Apply 3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000085" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "relace/relace-apply-3", "canonicalSlug": "relace/relace-apply-3", "servingProvider": "Relace", "servingProviderSlug": "relace", "contextLength": 256000, "maxCompletionTokens": 128000, "quantization": "fp8", "status": 0, "supportedParameters": [ "stop", "seed", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Relace | relace/relace-apply-3", "model_id": "relace/relace-apply-3", "model_name": "Relace: Relace Apply 3", "context_length": 256000, "pricing": { "prompt": "0.00000085", "completion": "0.00000125", "discount": 0 }, "provider_name": "Relace", "tag": "relace/fp8", "quantization": "fp8", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "stop", "seed", "max_tokens" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-235b-a22b-thinking@alibaba", "name": "Qwen: Qwen3 VL 235B A22B Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-235b-a22b-thinking", "canonicalSlug": "qwen/qwen3-vl-235b-a22b-thinking", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-vl-235b-a22b-thinking", "model_id": "qwen/qwen3-vl-235b-a22b-thinking", "model_name": "Qwen: Qwen3 VL 235B A22B Thinking", "context_length": 131072, "pricing": { "prompt": "0.0000004", "completion": "0.000004", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 126976, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 86.2529590801488, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-235b-a22b-thinking@novita", "name": "Qwen: Qwen3 VL 235B A22B Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.98, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000098" } ], "output": [ { "amount": 3.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000395" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-235b-a22b-thinking", "canonicalSlug": "qwen/qwen3-vl-235b-a22b-thinking", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3-vl-235b-a22b-thinking", "model_id": "qwen/qwen3-vl-235b-a22b-thinking", "model_name": "Qwen: Qwen3 VL 235B A22B Thinking", "context_length": 131072, "pricing": { "prompt": "0.00000098", "completion": "0.00000395", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.4006309148265, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-235b-a22b-instruct@deepinfra", "name": "Qwen: Qwen3 VL 235B A22B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000088" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-235b-a22b-instruct", "canonicalSlug": "qwen/qwen3-vl-235b-a22b-instruct", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...", "uptimeLast30m": 99.77309562398703, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-vl-235b-a22b-instruct", "model_id": "qwen/qwen3-vl-235b-a22b-instruct", "model_name": "Qwen: Qwen3 VL 235B A22B Instruct", "context_length": 262144, "pricing": { "prompt": "0.0000002", "completion": "0.00000088", "input_cache_read": "0.00000011", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.77309562398703, "uptime_last_5m": 98.8009592326139, "uptime_last_1d": 97.02288906535156, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-235b-a22b-instruct@venice", "name": "Qwen: Qwen3 VL 235B A22B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000021" } ], "output": [ { "amount": 1.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000019" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-235b-a22b-instruct", "canonicalSlug": "qwen/qwen3-vl-235b-a22b-instruct", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...", "uptimeLast30m": 99.15966386554622, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | qwen/qwen3-vl-235b-a22b-instruct", "model_id": "qwen/qwen3-vl-235b-a22b-instruct", "model_name": "Qwen: Qwen3 VL 235B A22B Instruct", "context_length": 128000, "pricing": { "prompt": "0.00000021", "completion": "0.0000019", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.15966386554622, "uptime_last_5m": null, "uptime_last_1d": 95.93656184613518, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-235b-a22b-instruct@parasail", "name": "Qwen: Qwen3 VL 235B A22B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000021" } ], "output": [ { "amount": 1.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000019" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-235b-a22b-instruct", "canonicalSlug": "qwen/qwen3-vl-235b-a22b-instruct", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...", "uptimeLast30m": 99.84779299847793, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | qwen/qwen3-vl-235b-a22b-instruct", "model_id": "qwen/qwen3-vl-235b-a22b-instruct", "model_name": "Qwen: Qwen3 VL 235B A22B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000021", "completion": "0.0000019", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.84779299847793, "uptime_last_5m": 99.835255354201, "uptime_last_1d": 99.44869344564974, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-235b-a22b-instruct@alibaba", "name": "Qwen: Qwen3 VL 235B A22B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "output": [ { "amount": 1.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000104" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-235b-a22b-instruct", "canonicalSlug": "qwen/qwen3-vl-235b-a22b-instruct", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-vl-235b-a22b-instruct", "model_id": "qwen/qwen3-vl-235b-a22b-instruct", "model_name": "Qwen: Qwen3 VL 235B A22B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000026", "completion": "0.00000104", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 129024, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.93666501731816, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-vl-235b-a22b-instruct@novita", "name": "Qwen: Qwen3 VL 235B A22B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-vl-235b-a22b-instruct", "canonicalSlug": "qwen/qwen3-vl-235b-a22b-instruct", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "bf16", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...", "uptimeLast30m": 95.8477508650519, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3-vl-235b-a22b-instruct", "model_id": "qwen/qwen3-vl-235b-a22b-instruct", "model_name": "Qwen: Qwen3 VL 235B A22B Instruct", "context_length": 131072, "pricing": { "prompt": "0.0000003", "completion": "0.0000015", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 95.8477508650519, "uptime_last_5m": 95, "uptime_last_1d": 93.83911625869872, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-max@alibaba", "name": "Qwen: Qwen3 Max", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.78, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000078" } ], "output": [ { "amount": 3.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000039" } ], "cacheRead": [ { "amount": 0.156, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000156" } ], "cacheWrite": [ { "amount": 0.975, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000975" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-max", "canonicalSlug": "qwen/qwen3-max", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-max", "model_id": "qwen/qwen3-max", "model_name": "Qwen: Qwen3 Max", "context_length": 262144, "pricing": { "prompt": "0.00000078", "completion": "0.0000039", "input_cache_read": "0.000000156", "input_cache_write": "0.000000975", "discount": 0, "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.00000156", "completion": "0.0000078", "input_cache_read": "0.000000312", "input_cache_write": "0.00000195" }, { "min_prompt_tokens": 128000, "prompt": "0.00000195", "completion": "0.00000975", "input_cache_read": "0.00000039", "input_cache_write": "0.0000024375" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 258048, "supported_parameters": [ "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder-plus@alibaba", "name": "Qwen: Qwen3 Coder Plus", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000065" } ], "output": [ { "amount": 3.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000325" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheWrite": [ { "amount": 0.8125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008125" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder-plus", "canonicalSlug": "qwen/qwen3-coder-plus", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-coder-plus", "model_id": "qwen/qwen3-coder-plus", "model_name": "Qwen: Qwen3 Coder Plus", "context_length": 1000000, "pricing": { "prompt": "0.00000065", "completion": "0.00000325", "input_cache_read": "0.00000013", "input_cache_write": "0.0000008125", "discount": 0, "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.00000117", "completion": "0.00000585", "input_cache_read": "0.000000234", "input_cache_write": "0.0000014625" }, { "min_prompt_tokens": 128000, "prompt": "0.00000195", "completion": "0.00000975", "input_cache_read": "0.00000039", "input_cache_write": "0.0000024375" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 997952, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-codex:batch@openai", "name": "OpenAI: GPT-5 Codex", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-codex:batch", "canonicalSlug": "openai/gpt-5-codex", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-codex:batch", "model_id": "openai/gpt-5-codex:batch", "model_name": "OpenAI: GPT-5 Codex", "context_length": 400000, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000000625", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.1-terminus@siliconflow", "name": "DeepSeek: DeepSeek V3.1 Terminus", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.1-terminus", "canonicalSlug": "deepseek/deepseek-v3.1-terminus", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 163840, "maxCompletionTokens": 163840, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...", "uptimeLast30m": 98.03625377643505, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | deepseek/deepseek-v3.1-terminus", "model_id": "deepseek/deepseek-v3.1-terminus", "model_name": "DeepSeek: DeepSeek V3.1 Terminus", "context_length": 163840, "pricing": { "prompt": "0.00000027", "completion": "0.000001", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 163840, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 98.03625377643505, "uptime_last_5m": 99.15966386554622, "uptime_last_1d": 97.41520043533465, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.1-terminus@novita", "name": "DeepSeek: DeepSeek V3.1 Terminus", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000135" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.1-terminus", "canonicalSlug": "deepseek/deepseek-v3.1-terminus", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-v3.1-terminus", "model_id": "deepseek/deepseek-v3.1-terminus", "model_name": "DeepSeek: DeepSeek V3.1 Terminus", "context_length": 131072, "pricing": { "prompt": "0.00000027", "completion": "0.000001", "input_cache_read": "0.000000135", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.97766195085629, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.1-terminus@atlascloud", "name": "DeepSeek: DeepSeek V3.1 Terminus", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.1-terminus", "canonicalSlug": "deepseek/deepseek-v3.1-terminus", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 131072, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | deepseek/deepseek-v3.1-terminus", "model_id": "deepseek/deepseek-v3.1-terminus", "model_name": "DeepSeek: DeepSeek V3.1 Terminus", "context_length": 131072, "pricing": { "prompt": "0.0000003", "completion": "0.00000095", "input_cache_read": "0.00000013", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.26903360546801, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-v3.1-terminus@streamlake", "name": "DeepSeek: DeepSeek V3.1 Terminus", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3426, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003426" } ], "output": [ { "amount": 1.0284, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000010284" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-v3.1-terminus", "canonicalSlug": "deepseek/deepseek-v3.1-terminus", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 128000, "maxCompletionTokens": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | deepseek/deepseek-v3.1-terminus", "model_id": "deepseek/deepseek-v3.1-terminus", "model_name": "DeepSeek: DeepSeek V3.1 Terminus", "context_length": 128000, "pricing": { "prompt": "0.0000003426", "completion": "0.0000010284", "discount": 0.4 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.8884024379775, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder-flash@alibaba", "name": "Qwen: Qwen3 Coder Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.195, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000195" } ], "output": [ { "amount": 0.975, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000975" } ], "cacheRead": [ { "amount": 0.039, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000039" } ], "cacheWrite": [ { "amount": 0.24375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000024375" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder-flash", "canonicalSlug": "qwen/qwen3-coder-flash", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-coder-flash", "model_id": "qwen/qwen3-coder-flash", "model_name": "Qwen: Qwen3 Coder Flash", "context_length": 1000000, "pricing": { "prompt": "0.000000195", "completion": "0.000000975", "input_cache_read": "0.000000039", "input_cache_write": "0.00000024375", "discount": 0, "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.000000325", "completion": "0.000001625", "input_cache_read": "0.000000065", "input_cache_write": "0.00000040625" }, { "min_prompt_tokens": 128000, "prompt": "0.00000052", "completion": "0.0000026", "input_cache_read": "0.000000104", "input_cache_write": "0.00000065" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 997952, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-next-80b-a3b-thinking@google", "name": "Qwen: Qwen3 Next 80B A3B Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-next-80b-a3b-thinking", "canonicalSlug": "qwen/qwen3-next-80b-a3b-thinking-2509", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | qwen/qwen3-next-80b-a3b-thinking-2509", "model_id": "qwen/qwen3-next-80b-a3b-thinking", "model_name": "Qwen: Qwen3 Next 80B A3B Thinking", "context_length": 262144, "pricing": { "prompt": "0.00000015", "completion": "0.0000012", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-next-80b-a3b-thinking@nebius", "name": "Qwen: Qwen3 Next 80B A3B Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-next-80b-a3b-thinking", "canonicalSlug": "qwen/qwen3-next-80b-a3b-thinking-2509", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 8000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "repetition_penalty", "response_format", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | qwen/qwen3-next-80b-a3b-thinking-2509", "model_id": "qwen/qwen3-next-80b-a3b-thinking", "model_name": "Qwen: Qwen3 Next 80B A3B Thinking", "context_length": 8000, "pricing": { "prompt": "0.00000015", "completion": "0.0000012", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "repetition_penalty", "response_format", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-next-80b-a3b-thinking@alibaba", "name": "Qwen: Qwen3 Next 80B A3B Thinking", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-next-80b-a3b-thinking", "canonicalSlug": "qwen/qwen3-next-80b-a3b-thinking-2509", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-next-80b-a3b-thinking-2509", "model_id": "qwen/qwen3-next-80b-a3b-thinking", "model_name": "Qwen: Qwen3 Next 80B A3B Thinking", "context_length": 131072, "pricing": { "prompt": "0.00000015", "completion": "0.0000012", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 126976, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.92810927390366, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-next-80b-a3b-instruct@deepinfra", "name": "Qwen: Qwen3 Next 80B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-next-80b-a3b-instruct", "canonicalSlug": "qwen/qwen3-next-80b-a3b-instruct-2509", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-next-80b-a3b-instruct-2509", "model_id": "qwen/qwen3-next-80b-a3b-instruct", "model_name": "Qwen: Qwen3 Next 80B A3B Instruct", "context_length": 262144, "pricing": { "prompt": "0.00000009", "completion": "0.0000011", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.22117187798997, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-next-80b-a3b-instruct@alibaba", "name": "Qwen: Qwen3 Next 80B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.0975, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000975" } ], "output": [ { "amount": 0.78, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000078" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-next-80b-a3b-instruct", "canonicalSlug": "qwen/qwen3-next-80b-a3b-instruct-2509", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-next-80b-a3b-instruct-2509", "model_id": "qwen/qwen3-next-80b-a3b-instruct", "model_name": "Qwen: Qwen3 Next 80B A3B Instruct", "context_length": 131072, "pricing": { "prompt": "0.0000000975", "completion": "0.00000078", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 129024, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99103380256433, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-next-80b-a3b-instruct@parasail", "name": "Qwen: Qwen3 Next 80B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-next-80b-a3b-instruct", "canonicalSlug": "qwen/qwen3-next-80b-a3b-instruct-2509", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | qwen/qwen3-next-80b-a3b-instruct-2509", "model_id": "qwen/qwen3-next-80b-a3b-instruct", "model_name": "Qwen: Qwen3 Next 80B A3B Instruct", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.0000011", "input_cache_read": "0.00000007", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.98601525160208, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-next-80b-a3b-instruct@google", "name": "Qwen: Qwen3 Next 80B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-next-80b-a3b-instruct", "canonicalSlug": "qwen/qwen3-next-80b-a3b-instruct-2509", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | qwen/qwen3-next-80b-a3b-instruct-2509", "model_id": "qwen/qwen3-next-80b-a3b-instruct", "model_name": "Qwen: Qwen3 Next 80B A3B Instruct", "context_length": 262144, "pricing": { "prompt": "0.00000015", "completion": "0.0000012", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96895924792693, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-next-80b-a3b-instruct@novita", "name": "Qwen: Qwen3 Next 80B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-next-80b-a3b-instruct", "canonicalSlug": "qwen/qwen3-next-80b-a3b-instruct-2509", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...", "uptimeLast30m": 99.38080495356037, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3-next-80b-a3b-instruct-2509", "model_id": "qwen/qwen3-next-80b-a3b-instruct", "model_name": "Qwen: Qwen3 Next 80B A3B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000015", "completion": "0.0000015", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.38080495356037, "uptime_last_5m": 100, "uptime_last_1d": 99.83550095101012, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-plus-2025-07-28@alibaba", "name": "Qwen: Qwen Plus 0728", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "output": [ { "amount": 0.78, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000078" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-plus-2025-07-28", "canonicalSlug": "qwen/qwen-plus-2025-07-28", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen-plus-2025-07-28", "model_id": "qwen/qwen-plus-2025-07-28", "model_name": "Qwen: Qwen Plus 0728", "context_length": 1000000, "pricing": { "prompt": "0.00000026", "completion": "0.00000078", "discount": 0, "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.00000078", "completion": "0.00000234" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 995904, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-plus-2025-07-28:thinking@alibaba", "name": "Qwen: Qwen Plus 0728", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "output": [ { "amount": 0.78, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000078" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-plus-2025-07-28:thinking", "canonicalSlug": "qwen/qwen-plus-2025-07-28", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen-plus-2025-07-28:thinking", "model_id": "qwen/qwen-plus-2025-07-28:thinking", "model_name": "Qwen: Qwen Plus 0728", "context_length": 1000000, "pricing": { "prompt": "0.00000026", "completion": "0.00000078", "discount": 0, "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.00000078", "completion": "0.00000234" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 995904, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nvidia/nemotron-nano-9b-v2:free@nvidia", "name": "NVIDIA: Nemotron Nano 9B V2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nvidia/nemotron-nano-9b-v2:free", "canonicalSlug": "nvidia/nemotron-nano-9b-v2", "servingProvider": "Nvidia", "servingProviderSlug": "nvidia", "contextLength": 128000, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "max_tokens", "seed", "top_p", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "NVIDIA-Nemotron-Nano-9B-v2 is a large language model (LLM) trained from scratch by NVIDIA, and designed as a unified model for both reasoning and non-reasoning tasks. It responds to user queries and...", "uptimeLast30m": 97.6659038901602, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nvidia | nvidia/nemotron-nano-9b-v2:free", "model_id": "nvidia/nemotron-nano-9b-v2:free", "model_name": "NVIDIA: Nemotron Nano 9B V2", "context_length": 128000, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Nvidia", "tag": "nvidia/bf16", "quantization": "bf16", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "max_tokens", "seed", "top_p", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 97.6659038901602, "uptime_last_5m": 98.94291754756871, "uptime_last_1d": 96.96844159054918, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2-0905@novita", "name": "MoonshotAI: Kimi K2 0905", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2-0905", "canonicalSlug": "moonshotai/kimi-k2-0905", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 100352, "quantization": "fp8", "status": 0, "supportedParameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | moonshotai/kimi-k2-0905", "model_id": "moonshotai/kimi-k2-0905", "model_name": "MoonshotAI: Kimi K2 0905", "context_length": 262144, "pricing": { "prompt": "0.0000006", "completion": "0.0000025", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 100352, "max_prompt_tokens": null, "supported_parameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9925845179566, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-30b-a3b-thinking-2507@alibaba", "name": "Qwen: Qwen3 30B A3B Thinking 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-30b-a3b-thinking-2507", "canonicalSlug": "qwen/qwen3-30b-a3b-thinking-2507", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 81920, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-30b-a3b-thinking-2507", "model_id": "qwen/qwen3-30b-a3b-thinking-2507", "model_name": "Qwen: Qwen3 30B A3B Thinking 2507", "context_length": 81920, "pricing": { "prompt": "0.0000002", "completion": "0.0000024", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 126976, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nousresearch/hermes-4-70b@nebius", "name": "Nous: Hermes 4 70B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nousresearch/hermes-4-70b", "canonicalSlug": "nousresearch/hermes-4-70b", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": null }, "description": "Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | nousresearch/hermes-4-70b", "model_id": "nousresearch/hermes-4-70b", "model_name": "Nous: Hermes 4 70B", "context_length": 131072, "pricing": { "prompt": "0.00000013", "completion": "0.0000004", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99581178145876, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nousresearch/hermes-4-405b@nebius", "name": "Nous: Hermes 4 405B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nousresearch/hermes-4-405b", "canonicalSlug": "nousresearch/hermes-4-405b", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | nousresearch/hermes-4-405b", "model_id": "nousresearch/hermes-4-405b", "model_name": "Nous: Hermes 4 405B", "context_length": 131072, "pricing": { "prompt": "0.000001", "completion": "0.000003", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3.1@deepinfra", "name": "DeepSeek: DeepSeek V3.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3.1", "canonicalSlug": "deepseek/deepseek-chat-v3.1", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 163840, "maxCompletionTokens": 32768, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...", "uptimeLast30m": 99.42376950780312, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | deepseek/deepseek-chat-v3.1", "model_id": "deepseek/deepseek-chat-v3.1", "model_name": "DeepSeek: DeepSeek V3.1", "context_length": 163840, "pricing": { "prompt": "0.00000025", "completion": "0.00000095", "input_cache_read": "0.00000013", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.42376950780312, "uptime_last_5m": 99.4, "uptime_last_1d": 99.63923805196382, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3.1@novita", "name": "DeepSeek: DeepSeek V3.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000135" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3.1", "canonicalSlug": "deepseek/deepseek-chat-v3.1", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-chat-v3.1", "model_id": "deepseek/deepseek-chat-v3.1", "model_name": "DeepSeek: DeepSeek V3.1", "context_length": 131072, "pricing": { "prompt": "0.00000027", "completion": "0.000001", "input_cache_read": "0.000000135", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.97635104830096, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3.1@siliconflow", "name": "DeepSeek: DeepSeek V3.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3.1", "canonicalSlug": "deepseek/deepseek-chat-v3.1", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 163840, "maxCompletionTokens": 163840, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...", "uptimeLast30m": 98.90981169474728, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | deepseek/deepseek-chat-v3.1", "model_id": "deepseek/deepseek-chat-v3.1", "model_name": "DeepSeek: DeepSeek V3.1", "context_length": 163840, "pricing": { "prompt": "0.00000027", "completion": "0.000001", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 163840, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 98.90981169474728, "uptime_last_5m": 99.09909909909909, "uptime_last_1d": 97.41405945648232, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3.1@atlascloud", "name": "DeepSeek: DeepSeek V3.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3.1", "canonicalSlug": "deepseek/deepseek-chat-v3.1", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 131072, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | deepseek/deepseek-chat-v3.1", "model_id": "deepseek/deepseek-chat-v3.1", "model_name": "DeepSeek: DeepSeek V3.1", "context_length": 131072, "pricing": { "prompt": "0.0000003", "completion": "0.00000095", "input_cache_read": "0.00000013", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 97.64018130624895, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3.1@coreweave", "name": "DeepSeek: DeepSeek V3.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "output": [ { "amount": 1.6500000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000165" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3.1", "canonicalSlug": "deepseek/deepseek-chat-v3.1", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 161000, "maxCompletionTokens": 161000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...", "uptimeLast30m": 99.98334998334998, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | deepseek/deepseek-chat-v3.1", "model_id": "deepseek/deepseek-chat-v3.1", "model_name": "DeepSeek: DeepSeek V3.1", "context_length": 161000, "pricing": { "prompt": "0.00000055", "completion": "0.00000165", "input_cache_read": "0.00000055", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp8", "quantization": "fp8", "max_completion_tokens": 161000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.98334998334998, "uptime_last_5m": 100, "uptime_last_1d": 99.9276813127646, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3.1@google", "name": "DeepSeek: DeepSeek V3.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 1.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000017" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3.1", "canonicalSlug": "deepseek/deepseek-chat-v3.1", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 163840, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | deepseek/deepseek-chat-v3.1", "model_id": "deepseek/deepseek-chat-v3.1", "model_name": "DeepSeek: DeepSeek V3.1", "context_length": 163840, "pricing": { "prompt": "0.0000006", "completion": "0.0000017", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us-west2", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 97.94732370433304, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3.1@mara", "name": "DeepSeek: DeepSeek V3.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 1.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000017" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3.1", "canonicalSlug": "deepseek/deepseek-chat-v3.1", "servingProvider": "Mara", "servingProviderSlug": "mara", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "top_k", "top_p", "response_format", "logprobs", "top_logprobs", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mara | deepseek/deepseek-chat-v3.1", "model_id": "deepseek/deepseek-chat-v3.1", "model_name": "DeepSeek: DeepSeek V3.1", "context_length": 131072, "pricing": { "prompt": "0.0000006", "completion": "0.0000017", "discount": 0 }, "provider_name": "Mara", "tag": "mara", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "top_k", "top_p", "response_format", "logprobs", "top_logprobs", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 93.42930893370549, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3.1@sambanova", "name": "DeepSeek: DeepSeek V3.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000065" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3.1", "canonicalSlug": "deepseek/deepseek-chat-v3.1", "servingProvider": "SambaNova", "servingProviderSlug": "sambanova", "contextLength": 131072, "maxCompletionTokens": 7168, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-v3.1" }, "description": "DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SambaNova | deepseek/deepseek-chat-v3.1", "model_id": "deepseek/deepseek-chat-v3.1", "model_name": "DeepSeek: DeepSeek V3.1", "context_length": 131072, "pricing": { "prompt": "0.00000065", "completion": "0.0000015", "discount": 0 }, "provider_name": "SambaNova", "tag": "sambanova/fp8", "quantization": "fp8", "max_completion_tokens": 7168, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 95.74167329269369, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-medium-3.1@mistral", "name": "Mistral: Mistral Medium 3.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-medium-3.1", "canonicalSlug": "mistralai/mistral-medium-3.1", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/mistral-medium-3.1", "model_id": "mistralai/mistral-medium-3.1", "model_name": "Mistral: Mistral Medium 3.1", "context_length": 131072, "pricing": { "prompt": "0.0000004", "completion": "0.000002", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.94480877865074, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.5v@novita", "name": "Z.ai: GLM 4.5V", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.5v", "canonicalSlug": "z-ai/glm-4.5v", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 65536, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | z-ai/glm-4.5v", "model_id": "z-ai/glm-4.5v", "model_name": "Z.ai: GLM 4.5V", "context_length": 65536, "pricing": { "prompt": "0.0000006", "completion": "0.0000018", "input_cache_read": "0.00000011", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 98.71622447731997, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.5v@z.ai", "name": "Z.ai: GLM 4.5V", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.5v", "canonicalSlug": "z-ai/glm-4.5v", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 65536, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k", "response_format" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...", "uptimeLast30m": 99.1869918699187, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-4.5v", "model_id": "z-ai/glm-4.5v", "model_name": "Z.ai: GLM 4.5V", "context_length": 65536, "pricing": { "prompt": "0.0000006", "completion": "0.0000018", "input_cache_read": "0.00000011", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k", "response_format" ], "status": 0, "uptime_last_30m": 99.1869918699187, "uptime_last_5m": null, "uptime_last_1d": 97.98932384341637, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "ai21/jamba-large-1.7@ai21", "name": "AI21: Jamba Large 1.7", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "ai21/jamba-large-1.7", "canonicalSlug": "ai21/jamba-large-1.7", "servingProvider": "AI21", "servingProviderSlug": "ai21", "contextLength": 256000, "maxCompletionTokens": 4096, "quantization": "fp8", "status": 0, "supportedParameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AI21 | ai21/jamba-large-1.7", "model_id": "ai21/jamba-large-1.7", "model_name": "AI21: Jamba Large 1.7", "context_length": 256000, "pricing": { "prompt": "0.000002", "completion": "0.000008", "discount": 0 }, "provider_name": "AI21", "tag": "ai21/fp8", "quantization": "fp8", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5@azure", "name": "OpenAI: GPT-5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5", "canonicalSlug": "openai/gpt-5-2025-08-07", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5-2025-08-07", "model_id": "openai/gpt-5", "model_name": "OpenAI: GPT-5", "context_length": 400000, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.000000125", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.98765889176849, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5@openai", "name": "OpenAI: GPT-5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5", "canonicalSlug": "openai/gpt-5-2025-08-07", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-2025-08-07", "model_id": "openai/gpt-5", "model_name": "OpenAI: GPT-5", "context_length": 400000, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "web_search": "0.01", "input_cache_read": "0.000000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/default", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96775426910445, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5@azure", "name": "OpenAI: GPT-5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.1375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001375" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5", "canonicalSlug": "openai/gpt-5-2025-08-07", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5-2025-08-07", "model_id": "openai/gpt-5", "model_name": "OpenAI: GPT-5", "context_length": 400000, "pricing": { "prompt": "0.000001375", "completion": "0.000011", "web_search": "0.01", "input_cache_read": "0.0000001375", "discount": 0 }, "provider_name": "Azure", "tag": "azure/swedencentral", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5:batch@openai", "name": "OpenAI: GPT-5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5:batch", "canonicalSlug": "openai/gpt-5-2025-08-07", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-2025-08-07:batch", "model_id": "openai/gpt-5:batch", "model_name": "OpenAI: GPT-5", "context_length": 400000, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.0000000625", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-mini@openai", "name": "OpenAI: GPT-5 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-mini", "canonicalSlug": "openai/gpt-5-mini-2025-08-07", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....", "uptimeLast30m": 99.94533003395293, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-mini-2025-08-07", "model_id": "openai/gpt-5-mini", "model_name": "OpenAI: GPT-5 Mini", "context_length": 400000, "pricing": { "prompt": "0.00000025", "completion": "0.000002", "web_search": "0.01", "input_cache_read": "0.000000025", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.94533003395293, "uptime_last_5m": 100, "uptime_last_1d": 99.95203211323788, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-mini@openai", "name": "OpenAI: GPT-5 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.012499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-mini", "canonicalSlug": "openai/gpt-5-mini-2025-08-07", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....", "uptimeLast30m": 99.94533003395293, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-mini-2025-08-07", "model_id": "openai/gpt-5-mini", "model_name": "OpenAI: GPT-5 Mini", "context_length": 400000, "pricing": { "prompt": "0.000000125", "completion": "0.000001", "web_search": "0.01", "input_cache_read": "0.0000000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.94533003395293, "uptime_last_5m": 100, "uptime_last_1d": 99.95203211323788, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-mini@azure", "name": "OpenAI: GPT-5 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-mini", "canonicalSlug": "openai/gpt-5-mini-2025-08-07", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5-mini-2025-08-07", "model_id": "openai/gpt-5-mini", "model_name": "OpenAI: GPT-5 Mini", "context_length": 400000, "pricing": { "prompt": "0.00000025", "completion": "0.000002", "web_search": "0.01", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99494954596419, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-mini@azure", "name": "OpenAI: GPT-5 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.032999999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000033" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-mini", "canonicalSlug": "openai/gpt-5-mini-2025-08-07", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5-mini-2025-08-07", "model_id": "openai/gpt-5-mini", "model_name": "OpenAI: GPT-5 Mini", "context_length": 400000, "pricing": { "prompt": "0.000000275", "completion": "0.0000022", "web_search": "0.01", "input_cache_read": "0.000000033", "discount": 0 }, "provider_name": "Azure", "tag": "azure/swedencentral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-mini:batch@openai", "name": "OpenAI: GPT-5 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.012499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-mini:batch", "canonicalSlug": "openai/gpt-5-mini-2025-08-07", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-mini-2025-08-07:batch", "model_id": "openai/gpt-5-mini:batch", "model_name": "OpenAI: GPT-5 Mini", "context_length": 400000, "pricing": { "prompt": "0.000000125", "completion": "0.000001", "web_search": "0.01", "input_cache_read": "0.0000000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-mini:batch@openai", "name": "OpenAI: GPT-5 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheRead": [ { "amount": 0.0062499999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000000625" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-mini:batch", "canonicalSlug": "openai/gpt-5-mini-2025-08-07", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-mini-2025-08-07:batch", "model_id": "openai/gpt-5-mini:batch", "model_name": "OpenAI: GPT-5 Mini", "context_length": 400000, "pricing": { "prompt": "0.0000000625", "completion": "0.0000005", "web_search": "0.01", "input_cache_read": "0.00000000625", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-nano@openai", "name": "OpenAI: GPT-5 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-nano", "canonicalSlug": "openai/gpt-5-nano-2025-08-07", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...", "uptimeLast30m": 99.14005652775272, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-nano-2025-08-07", "model_id": "openai/gpt-5-nano", "model_name": "OpenAI: GPT-5 Nano", "context_length": 400000, "pricing": { "prompt": "0.00000005", "completion": "0.0000004", "web_search": "0.01", "input_cache_read": "0.000000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.14005652775272, "uptime_last_5m": 99.52181709503886, "uptime_last_1d": 97.32466200682961, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-nano@openai", "name": "OpenAI: GPT-5 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [ { "amount": 0.0025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-nano", "canonicalSlug": "openai/gpt-5-nano-2025-08-07", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...", "uptimeLast30m": 99.14005652775272, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-nano-2025-08-07", "model_id": "openai/gpt-5-nano", "model_name": "OpenAI: GPT-5 Nano", "context_length": 400000, "pricing": { "prompt": "0.000000025", "completion": "0.0000002", "web_search": "0.01", "input_cache_read": "0.0000000025", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai/flex", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": 272000, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.14005652775272, "uptime_last_5m": 99.52181709503886, "uptime_last_1d": 97.32466200682961, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-nano@azure", "name": "OpenAI: GPT-5 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-nano", "canonicalSlug": "openai/gpt-5-nano-2025-08-07", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5-nano-2025-08-07", "model_id": "openai/gpt-5-nano", "model_name": "OpenAI: GPT-5 Nano", "context_length": 400000, "pricing": { "prompt": "0.00000005", "completion": "0.0000004", "web_search": "0.01", "input_cache_read": "0.00000001", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-nano@azure", "name": "OpenAI: GPT-5 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000055" } ], "output": [ { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000044" } ], "cacheRead": [ { "amount": 0.011, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000011" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-nano", "canonicalSlug": "openai/gpt-5-nano-2025-08-07", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 400000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-5-nano-2025-08-07", "model_id": "openai/gpt-5-nano", "model_name": "OpenAI: GPT-5 Nano", "context_length": 400000, "pricing": { "prompt": "0.000000055", "completion": "0.00000044", "web_search": "0.01", "input_cache_read": "0.000000011", "discount": 0 }, "provider_name": "Azure", "tag": "azure/swedencentral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "reasoning", "include_reasoning", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-5-nano:batch@openai", "name": "OpenAI: GPT-5 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [ { "amount": 0.0025, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-5-nano:batch", "canonicalSlug": "openai/gpt-5-nano-2025-08-07", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 400000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-5-nano-2025-08-07:batch", "model_id": "openai/gpt-5-nano:batch", "model_name": "OpenAI: GPT-5 Nano", "context_length": 400000, "pricing": { "prompt": "0.000000025", "completion": "0.0000002", "web_search": "0.01", "input_cache_read": "0.0000000025", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@coreweave", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "output": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000017" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 96.40724295989564, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000003", "completion": "0.00000017", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp4", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 96.40724295989564, "uptime_last_5m": 94.67524115755627, "uptime_last_1d": 96.88923225830852, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@deepinfra", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.037, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000037" } ], "output": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000017" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.4805965319166, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.000000037", "completion": "0.00000017", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.4805965319166, "uptime_last_5m": 99.4738819992484, "uptime_last_1d": 98.834061323853, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@akashml", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.037, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000037" } ], "output": [ { "amount": 0.49, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000049" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "AkashML", "servingProviderSlug": "akashml", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "stop", "seed", "frequency_penalty", "presence_penalty", "repetition_penalty", "max_tokens", "response_format", "structured_outputs", "tools", "logprobs", "top_logprobs", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.86429805082655, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AkashML | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.000000037", "completion": "0.00000049", "discount": 0 }, "provider_name": "AkashML", "tag": "akashml/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "stop", "seed", "frequency_penalty", "presence_penalty", "repetition_penalty", "max_tokens", "response_format", "structured_outputs", "tools", "logprobs", "top_logprobs", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.86429805082655, "uptime_last_5m": 100, "uptime_last_1d": 96.95797124981284, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@novita", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.22953451043338, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000005", "completion": "0.00000025", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp4", "quantization": "fp4", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.22953451043338, "uptime_last_5m": 100, "uptime_last_1d": 96.6339331671427, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@siliconflow", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 131072, "maxCompletionTokens": 8192, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.30039352864014, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000005", "completion": "0.00000045", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.30039352864014, "uptime_last_5m": 98.33333333333333, "uptime_last_1d": 90.30587912202431, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@digitalocean", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000055" } ], "output": [ { "amount": 0.385, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000385" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.97084123050007, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 128000, "pricing": { "prompt": "0.000000055", "completion": "0.000000385", "input_cache_read": "0.00000002", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "max_tokens", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.97084123050007, "uptime_last_5m": 99.8859749144812, "uptime_last_1d": 99.9052591688071, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@mancer-2", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000085" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Mancer 2", "servingProviderSlug": "mancer-2", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "response_format", "logprobs", "top_logprobs", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.82014388489209, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mancer 2 | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.000000085", "completion": "0.0000005", "discount": 0 }, "provider_name": "Mancer 2", "tag": "mancer/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "response_format", "logprobs", "top_logprobs", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.82014388489209, "uptime_last_5m": 98.75776397515527, "uptime_last_1d": 99.55979624420284, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@google", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "output": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "repetition_penalty", "top_k", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 94.50527573163448, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000009", "completion": "0.00000036", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "repetition_penalty", "top_k", "reasoning_effort" ], "status": -2, "uptime_last_30m": 94.50527573163448, "uptime_last_5m": 94.87179487179486, "uptime_last_1d": 90.63917542415291, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@baseten", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "BaseTen", "servingProviderSlug": "baseten", "contextLength": 128072, "maxCompletionTokens": 128072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.3664833851303, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "BaseTen | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 128072, "pricing": { "prompt": "0.0000001", "completion": "0.0000005", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "BaseTen", "tag": "baseten/fp4", "quantization": "fp4", "max_completion_tokens": 128072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.3664833851303, "uptime_last_5m": 100, "uptime_last_1d": 98.85890637145104, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@parasail", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [ { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000055" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.98723023879454, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.0000001", "completion": "0.00000075", "input_cache_read": "0.000000055", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp4", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.98723023879454, "uptime_last_5m": 100, "uptime_last_1d": 99.48430332889242, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@sambanova", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000095" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "SambaNova", "servingProviderSlug": "sambanova", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 98.38783970520497, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SambaNova | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000014", "completion": "0.00000095", "discount": 0 }, "provider_name": "SambaNova", "tag": "sambanova", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.38783970520497, "uptime_last_5m": 83.33333333333334, "uptime_last_1d": 94.22020394510106, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@amazon-bedrock", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 98.85057471264368, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.85057471264368, "uptime_last_5m": 100, "uptime_last_1d": 99.53257922782089, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@deepinfra", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "tools", "tool_choice", "logit_bias", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 98.59154929577466, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/turbo", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "tools", "tool_choice", "logit_bias", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 98.59154929577466, "uptime_last_5m": 97.91666666666666, "uptime_last_1d": 99.41233546308331, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@together", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "response_format", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice", "response_format", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 97.79001695910247, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@nebius", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "repetition_penalty", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.09407665505226, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/fp4", "quantization": "fp4", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "repetition_penalty", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.09407665505226, "uptime_last_5m": 97.2809667673716, "uptime_last_1d": 98.66465748632703, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@amazon-bedrock", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@phala", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.91926952842269, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@groq", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Groq", "servingProviderSlug": "groq", "contextLength": 131072, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.92439681809216, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Groq | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.000000075", "discount": 0 }, "provider_name": "Groq", "tag": "groq", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.92439681809216, "uptime_last_5m": 99.7978981406629, "uptime_last_1d": 99.9183535341841, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@mara", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Mara", "servingProviderSlug": "mara", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "top_k", "top_p", "tool_choice", "tools", "structured_outputs", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.18962722852513, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mara | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000015", "completion": "0.00000075", "discount": 0 }, "provider_name": "Mara", "tag": "mara", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "stop", "top_k", "top_p", "tool_choice", "tools", "structured_outputs", "response_format", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.18962722852513, "uptime_last_5m": 100, "uptime_last_1d": 99.05751345298454, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-120b@cerebras", "name": "OpenAI: gpt-oss-120b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-120b", "canonicalSlug": "openai/gpt-oss-120b", "servingProvider": "Cerebras", "servingProviderSlug": "cerebras", "contextLength": 131072, "maxCompletionTokens": 40960, "quantization": "fp16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "response_format", "structured_outputs", "frequency_penalty", "presence_penalty", "logit_bias", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...", "uptimeLast30m": 99.99349889481212, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cerebras | openai/gpt-oss-120b", "model_id": "openai/gpt-oss-120b", "model_name": "OpenAI: gpt-oss-120b", "context_length": 131072, "pricing": { "prompt": "0.00000035", "completion": "0.00000075", "input_cache_read": "0.00000035", "discount": 0 }, "provider_name": "Cerebras", "tag": "cerebras/fp16", "quantization": "fp16", "max_completion_tokens": 40960, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "logprobs", "top_logprobs", "tools", "tool_choice", "response_format", "structured_outputs", "frequency_penalty", "presence_penalty", "logit_bias", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.99349889481212, "uptime_last_5m": 100, "uptime_last_1d": 99.98741823206637, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@coreweave", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "output": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": 99.718733465148, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.00000003", "completion": "0.00000013", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp4", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.718733465148, "uptime_last_5m": 99.63933637893724, "uptime_last_1d": 98.84894580215688, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@deepinfra", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": 99.81962864721486, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.00000003", "completion": "0.00000014", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.81962864721486, "uptime_last_5m": 100, "uptime_last_1d": 99.79375969537455, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@parasail", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": 99.14733657284313, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.00000003", "completion": "0.00000015", "input_cache_read": "0.00000002", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp4", "quantization": "fp4", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.14733657284313, "uptime_last_5m": 99.73941368078177, "uptime_last_1d": 96.98907987069305, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@novita", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.00000004", "completion": "0.00000015", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp4", "quantization": "fp4", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "structured_outputs", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.30085115824556, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@phala", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": 96.72750157331656, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.00000004", "completion": "0.00000015", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 96.72750157331656, "uptime_last_5m": 97.2972972972973, "uptime_last_1d": 93.40346886551038, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@siliconflow", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 131072, "maxCompletionTokens": 8192, "quantization": "fp8", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": 91.8918918918919, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.00000004", "completion": "0.00000018", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens", "reasoning_effort" ], "status": -2, "uptime_last_30m": 91.8918918918919, "uptime_last_5m": 46.478873239436616, "uptime_last_1d": 98.16520071878735, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@together", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.00000005", "completion": "0.0000002", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 95.74468085106383, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@amazon-bedrock", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.00000007", "completion": "0.00000015", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@amazon-bedrock", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": 99.97498123592695, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.00000007", "completion": "0.00000015", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.97498123592695, "uptime_last_5m": 100, "uptime_last_1d": 94.5137014552327, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@google", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": -5, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "repetition_penalty", "top_k", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": 78.02197802197803, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.00000007", "completion": "0.00000025", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us-central1", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "repetition_penalty", "top_k", "reasoning_effort" ], "status": -5, "uptime_last_30m": 78.02197802197803, "uptime_last_5m": 84.33734939759037, "uptime_last_1d": 85.65522286005056, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@fireworks", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000035" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "Fireworks", "servingProviderSlug": "fireworks", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": 99.36034115138592, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Fireworks | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.00000007", "completion": "0.0000003", "input_cache_read": "0.000000035", "discount": 0 }, "provider_name": "Fireworks", "tag": "fireworks", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": 99.36034115138592, "uptime_last_5m": 100, "uptime_last_1d": 97.68871023847254, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b@groq", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "Groq", "servingProviderSlug": "groq", "contextLength": 131072, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": 96.54019929442347, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Groq | openai/gpt-oss-20b", "model_id": "openai/gpt-oss-20b", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0.000000075", "completion": "0.0000003", "input_cache_read": "0.0000000375", "discount": 0 }, "provider_name": "Groq", "tag": "groq", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice", "structured_outputs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 96.54019929442347, "uptime_last_5m": 92.01497192763568, "uptime_last_1d": 98.33707396990444, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-oss-20b:free@darkbloom", "name": "OpenAI: gpt-oss-20b", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-oss-20b:free", "canonicalSlug": "openai/gpt-oss-20b", "servingProvider": "Darkbloom", "servingProviderSlug": "darkbloom", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "tools", "structured_outputs", "response_format", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...", "uptimeLast30m": 96.09015639374425, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Darkbloom | openai/gpt-oss-20b:free", "model_id": "openai/gpt-oss-20b:free", "model_name": "OpenAI: gpt-oss-20b", "context_length": 131072, "pricing": { "prompt": "0", "completion": "0", "discount": 0 }, "provider_name": "Darkbloom", "tag": "darkbloom", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "max_tokens", "tools", "structured_outputs", "response_format", "tool_choice", "logprobs", "top_logprobs", "reasoning_effort" ], "status": 0, "uptime_last_30m": 96.09015639374425, "uptime_last_5m": 97.79507133592736, "uptime_last_1d": 96.61596818024937, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.1@google", "name": "Anthropic: Claude Opus 4.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.1", "canonicalSlug": "anthropic/claude-4.1-opus-20250805", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 200000, "maxCompletionTokens": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4.1-opus-20250805", "model_id": "anthropic/claude-opus-4.1", "model_name": "Anthropic: Claude Opus 4.1", "context_length": 200000, "pricing": { "prompt": "0.000015", "completion": "0.000075", "web_search": "0.01", "input_cache_read": "0.0000015", "input_cache_write": "0.00001875", "input_cache_write_1h": "0.00003", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.1@amazon-bedrock", "name": "Anthropic: Claude Opus 4.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.1", "canonicalSlug": "anthropic/claude-4.1-opus-20250805", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 200000, "maxCompletionTokens": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4.1-opus-20250805", "model_id": "anthropic/claude-opus-4.1", "model_name": "Anthropic: Claude Opus 4.1", "context_length": 200000, "pricing": { "prompt": "0.000015", "completion": "0.000075", "web_search": "0.01", "input_cache_read": "0.0000015", "input_cache_write": "0.00001875", "input_cache_write_1h": "0.00003", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.8393456988462, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4.1:batch@anthropic", "name": "Anthropic: Claude Opus 4.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "output": [ { "amount": 37.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000375" } ], "cacheRead": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheWrite": [ { "amount": 9.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4.1:batch", "canonicalSlug": "anthropic/claude-4.1-opus-20250805", "servingProvider": "Anthropic", "servingProviderSlug": "anthropic", "contextLength": 200000, "maxCompletionTokens": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Anthropic | anthropic/claude-4.1-opus-20250805:batch", "model_id": "anthropic/claude-opus-4.1:batch", "model_name": "Anthropic: Claude Opus 4.1", "context_length": 200000, "pricing": { "prompt": "0.0000075", "completion": "0.0000375", "web_search": "0.01", "input_cache_read": "0.00000075", "input_cache_write": "0.000009375", "input_cache_write_1h": "0.000015", "discount": 0 }, "provider_name": "Anthropic", "tag": "anthropic", "quantization": "unknown", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/codestral-2508@mistral", "name": "Mistral: Codestral 2508", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/codestral-2508", "canonicalSlug": "mistralai/codestral-2508", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 256000, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "prediction", "tools", "tool_choice" ], "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/codestral-2508", "model_id": "mistralai/codestral-2508", "model_name": "Mistral: Codestral 2508", "context_length": 256000, "pricing": { "prompt": "0.0000003", "completion": "0.0000009", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "prediction", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.8340572990385, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder-30b-a3b-instruct@novita", "name": "Qwen: Qwen3 Coder 30B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "output": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder-30b-a3b-instruct", "canonicalSlug": "qwen/qwen3-coder-30b-a3b-instruct", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 160000, "maxCompletionTokens": 32768, "quantization": "fp8", "status": -2, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...", "uptimeLast30m": 89.4927536231884, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3-coder-30b-a3b-instruct", "model_id": "qwen/qwen3-coder-30b-a3b-instruct", "model_name": "Qwen: Qwen3 Coder 30B A3B Instruct", "context_length": 160000, "pricing": { "prompt": "0.00000007", "completion": "0.00000027", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "status": -2, "uptime_last_30m": 89.4927536231884, "uptime_last_5m": 89.47368421052632, "uptime_last_1d": 85.03747837785893, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder-30b-a3b-instruct@siliconflow", "name": "Qwen: Qwen3 Coder 30B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder-30b-a3b-instruct", "canonicalSlug": "qwen/qwen3-coder-30b-a3b-instruct", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...", "uptimeLast30m": 99.47916666666666, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3-coder-30b-a3b-instruct", "model_id": "qwen/qwen3-coder-30b-a3b-instruct", "model_name": "Qwen: Qwen3 Coder 30B A3B Instruct", "context_length": 262144, "pricing": { "prompt": "0.00000007", "completion": "0.00000028", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 99.47916666666666, "uptime_last_5m": 100, "uptime_last_1d": 95.31720545208022, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder-30b-a3b-instruct@amazon-bedrock", "name": "Qwen: Qwen3 Coder 30B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder-30b-a3b-instruct", "canonicalSlug": "qwen/qwen3-coder-30b-a3b-instruct", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 0, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tool_choice", "tools" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | qwen/qwen3-coder-30b-a3b-instruct", "model_id": "qwen/qwen3-coder-30b-a3b-instruct", "model_name": "Qwen: Qwen3 Coder 30B A3B Instruct", "context_length": 0, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tool_choice", "tools" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder-30b-a3b-instruct@alibaba", "name": "Qwen: Qwen3 Coder 30B A3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.2925, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002925" } ], "output": [ { "amount": 1.4625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014625" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder-30b-a3b-instruct", "canonicalSlug": "qwen/qwen3-coder-30b-a3b-instruct", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-coder-30b-a3b-instruct", "model_id": "qwen/qwen3-coder-30b-a3b-instruct", "model_name": "Qwen: Qwen3 Coder 30B A3B Instruct", "context_length": 262144, "pricing": { "prompt": "0.0000002925", "completion": "0.0000014625", "discount": 0, "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.0000004875", "completion": "0.0000024375" }, { "min_prompt_tokens": 128000, "prompt": "0.00000078", "completion": "0.0000039" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 204800, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.93961352657004, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-30b-a3b-instruct-2507@streamlake", "name": "Qwen: Qwen3 30B A3B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.04815, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004815" } ], "output": [ { "amount": 0.19305, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000019305" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-30b-a3b-instruct-2507", "canonicalSlug": "qwen/qwen3-30b-a3b-instruct-2507", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 128000, "maxCompletionTokens": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...", "uptimeLast30m": 99.92057188244638, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | qwen/qwen3-30b-a3b-instruct-2507", "model_id": "qwen/qwen3-30b-a3b-instruct-2507", "model_name": "Qwen: Qwen3 30B A3B Instruct 2507", "context_length": 128000, "pricing": { "prompt": "0.00000004815", "completion": "0.00000019305", "discount": 0.55 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "status": 0, "uptime_last_30m": 99.92057188244638, "uptime_last_5m": 100, "uptime_last_1d": 99.77238879046105, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-30b-a3b-instruct-2507@siliconflow", "name": "Qwen: Qwen3 30B A3B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-30b-a3b-instruct-2507", "canonicalSlug": "qwen/qwen3-30b-a3b-instruct-2507", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...", "uptimeLast30m": 99.4235239423524, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3-30b-a3b-instruct-2507", "model_id": "qwen/qwen3-30b-a3b-instruct-2507", "model_name": "Qwen: Qwen3 30B A3B Instruct 2507", "context_length": 262144, "pricing": { "prompt": "0.00000009", "completion": "0.0000003", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 99.4235239423524, "uptime_last_5m": 99.81369352585003, "uptime_last_1d": 88.75014017269277, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-30b-a3b-instruct-2507@nebius", "name": "Qwen: Qwen3 30B A3B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-30b-a3b-instruct-2507", "canonicalSlug": "qwen/qwen3-30b-a3b-instruct-2507", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "repetition_penalty" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...", "uptimeLast30m": 98.80260818020155, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | qwen/qwen3-30b-a3b-instruct-2507", "model_id": "qwen/qwen3-30b-a3b-instruct-2507", "model_name": "Qwen: Qwen3 30B A3B Instruct 2507", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.0000003", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "repetition_penalty" ], "status": 0, "uptime_last_30m": 98.80260818020155, "uptime_last_5m": 98.77697841726619, "uptime_last_1d": 97.9340366945492, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-30b-a3b-instruct-2507@coreweave", "name": "Qwen: Qwen3 30B A3B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-30b-a3b-instruct-2507", "canonicalSlug": "qwen/qwen3-30b-a3b-instruct-2507", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | qwen/qwen3-30b-a3b-instruct-2507", "model_id": "qwen/qwen3-30b-a3b-instruct-2507", "model_name": "Qwen: Qwen3 30B A3B Instruct 2507", "context_length": 262144, "pricing": { "prompt": "0.0000001", "completion": "0.0000003", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/bf16", "quantization": "bf16", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96955494355095, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-30b-a3b-instruct-2507@alibaba", "name": "Qwen: Qwen3 30B A3B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000052" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-30b-a3b-instruct-2507", "canonicalSlug": "qwen/qwen3-30b-a3b-instruct-2507", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...", "uptimeLast30m": 99.97427321842038, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-30b-a3b-instruct-2507", "model_id": "qwen/qwen3-30b-a3b-instruct-2507", "model_name": "Qwen: Qwen3 30B A3B Instruct 2507", "context_length": 131072, "pricing": { "prompt": "0.00000013", "completion": "0.00000052", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 129024, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 99.97427321842038, "uptime_last_5m": 100, "uptime_last_1d": 99.94826953242331, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.5@z.ai", "name": "Z.ai: GLM 4.5", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.5", "canonicalSlug": "z-ai/glm-4.5", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 131072, "maxCompletionTokens": 98304, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-4.5", "model_id": "z-ai/glm-4.5", "model_name": "Z.ai: GLM 4.5", "context_length": 131072, "pricing": { "prompt": "0.0000006", "completion": "0.0000022", "input_cache_read": "0.00000011", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai/fp8", "quantization": "fp8", "max_completion_tokens": 98304, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.5-air@novita", "name": "Z.ai: GLM 4.5 Air", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000085" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.5-air", "canonicalSlug": "z-ai/glm-4.5-air", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 98304, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...", "uptimeLast30m": 99.85949417904456, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | z-ai/glm-4.5-air", "model_id": "z-ai/glm-4.5-air", "model_name": "Z.ai: GLM 4.5 Air", "context_length": 131072, "pricing": { "prompt": "0.00000013", "completion": "0.00000085", "input_cache_read": "0.000000025", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 98304, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.85949417904456, "uptime_last_5m": 99.78991596638656, "uptime_last_1d": 99.62338016303674, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.5-air@siliconflow", "name": "Z.ai: GLM 4.5 Air", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.86, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000086" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.5-air", "canonicalSlug": "z-ai/glm-4.5-air", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp8", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...", "uptimeLast30m": 92.16354344122658, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | z-ai/glm-4.5-air", "model_id": "z-ai/glm-4.5-air", "model_name": "Z.ai: GLM 4.5 Air", "context_length": 131072, "pricing": { "prompt": "0.00000014", "completion": "0.00000086", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": -2, "uptime_last_30m": 92.16354344122658, "uptime_last_5m": 97.72727272727273, "uptime_last_1d": 99.11124133216134, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "z-ai/glm-4.5-air@z.ai", "name": "Z.ai: GLM 4.5 Air", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "z-ai/glm-4.5-air", "canonicalSlug": "z-ai/glm-4.5-air", "servingProvider": "Z.AI", "servingProviderSlug": "z.ai", "contextLength": 131072, "maxCompletionTokens": 98304, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...", "uptimeLast30m": 99.27007299270073, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Z.AI | z-ai/glm-4.5-air", "model_id": "z-ai/glm-4.5-air", "model_name": "Z.ai: GLM 4.5 Air", "context_length": 131072, "pricing": { "prompt": "0.0000002", "completion": "0.0000011", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Z.AI", "tag": "z-ai/fp8", "quantization": "fp8", "max_completion_tokens": 98304, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "tools", "tool_choice", "top_k" ], "status": 0, "uptime_last_30m": 99.27007299270073, "uptime_last_5m": null, "uptime_last_1d": 98.68980887418113, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-thinking-2507@deepinfra", "name": "Qwen: Qwen3 235B A22B Thinking 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000023" } ], "output": [ { "amount": 2.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000023" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-thinking-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-thinking-2507", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-235b-a22b-thinking-2507", "model_id": "qwen/qwen3-235b-a22b-thinking-2507", "model_name": "Qwen: Qwen3 235B A22B Thinking 2507", "context_length": 262144, "pricing": { "prompt": "0.00000023", "completion": "0.0000023", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.94514005421453, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-thinking-2507@alibaba", "name": "Qwen: Qwen3 235B A22B Thinking 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000023" } ], "output": [ { "amount": 2.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000023" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-thinking-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-thinking-2507", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-235b-a22b-thinking-2507", "model_id": "qwen/qwen3-235b-a22b-thinking-2507", "model_name": "Qwen: Qwen3 235B A22B Thinking 2507", "context_length": 131072, "pricing": { "prompt": "0.00000023", "completion": "0.0000023", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.94195132842152, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-thinking-2507@novita", "name": "Qwen: Qwen3 235B A22B Thinking 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-thinking-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-thinking-2507", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3-235b-a22b-thinking-2507", "model_id": "qwen/qwen3-235b-a22b-thinking-2507", "model_name": "Qwen: Qwen3 235B A22B Thinking 2507", "context_length": 131072, "pricing": { "prompt": "0.0000003", "completion": "0.000003", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.36654189461561, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-thinking-2507@venice", "name": "Qwen: Qwen3 235B A22B Thinking 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-thinking-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-thinking-2507", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | qwen/qwen3-235b-a22b-thinking-2507", "model_id": "qwen/qwen3-235b-a22b-thinking-2507", "model_name": "Qwen: Qwen3 235B A22B Thinking 2507", "context_length": 128000, "pricing": { "prompt": "0.00000045", "completion": "0.0000035", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.90965410428497, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder@google", "name": "Qwen: Qwen3 Coder 480B A35B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder", "canonicalSlug": "qwen/qwen3-coder-480b-a35b-07-25", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | qwen/qwen3-coder-480b-a35b-07-25", "model_id": "qwen/qwen3-coder", "model_name": "Qwen: Qwen3 Coder 480B A35B", "context_length": 262144, "pricing": { "prompt": "0.00000022", "completion": "0.0000018", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us-south1", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.20912197182017, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder@deepinfra", "name": "Qwen: Qwen3 Coder 480B A35B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder", "canonicalSlug": "qwen/qwen3-coder-480b-a35b-07-25", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp4", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...", "uptimeLast30m": 99.93891264508247, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-coder-480b-a35b-07-25", "model_id": "qwen/qwen3-coder", "model_name": "Qwen: Qwen3 Coder 480B A35B", "context_length": 262144, "pricing": { "prompt": "0.0000003", "completion": "0.000001", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/turbo", "quantization": "fp4", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.93891264508247, "uptime_last_5m": 100, "uptime_last_1d": 97.67720828789531, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder@venice", "name": "Qwen: Qwen3 Coder 480B A35B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder", "canonicalSlug": "qwen/qwen3-coder-480b-a35b-07-25", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | qwen/qwen3-coder-480b-a35b-07-25", "model_id": "qwen/qwen3-coder", "model_name": "Qwen: Qwen3 Coder 480B A35B", "context_length": 256000, "pricing": { "prompt": "0.00000035", "completion": "0.0000015", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 90.9601259181532, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder@novita", "name": "Qwen: Qwen3 Coder 480B A35B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000038" } ], "output": [ { "amount": 1.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000155" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder", "canonicalSlug": "qwen/qwen3-coder-480b-a35b-07-25", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3-coder-480b-a35b-07-25", "model_id": "qwen/qwen3-coder", "model_name": "Qwen: Qwen3 Coder 480B A35B", "context_length": 262144, "pricing": { "prompt": "0.00000038", "completion": "0.00000155", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 86.83922712358732, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-coder@alibaba", "name": "Qwen: Qwen3 Coder 480B A35B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.975, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000975" } ], "output": [ { "amount": 4.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004875" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-coder", "canonicalSlug": "qwen/qwen3-coder-480b-a35b-07-25", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 262144, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-coder-480b-a35b-07-25", "model_id": "qwen/qwen3-coder", "model_name": "Qwen: Qwen3 Coder 480B A35B", "context_length": 262144, "pricing": { "prompt": "0.000000975", "completion": "0.000004875", "discount": 0, "overrides": [ { "min_prompt_tokens": 32000, "prompt": "0.000001755", "completion": "0.000008775" }, { "min_prompt_tokens": 128000, "prompt": "0.000002925", "completion": "0.000014625" } ] }, "provider_name": "Alibaba", "tag": "alibaba/opensource", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": 204800, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "bytedance/ui-tars-1.5-7b@parasail", "name": "ByteDance: UI-TARS 7B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "bytedance/ui-tars-1.5-7b", "canonicalSlug": "bytedance/ui-tars-1.5-7b", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 128000, "maxCompletionTokens": 2048, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | bytedance/ui-tars-1.5-7b", "model_id": "bytedance/ui-tars-1.5-7b", "model_name": "ByteDance: UI-TARS 7B ", "context_length": 128000, "pricing": { "prompt": "0.0000001", "completion": "0.0000002", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 2048, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash-lite@google", "name": "Google: Gemini 2.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 1e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000001" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash-lite", "canonicalSlug": "google/gemini-2.5-flash-lite", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...", "uptimeLast30m": 99.93330512060939, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-flash-lite", "model_id": "google/gemini-2.5-flash-lite", "model_name": "Google: Gemini 2.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.0000001", "completion": "0.0000004", "image": "0.0000001", "audio": "0.0000003", "input_audio_cache": "0.00000003", "web_search": "0.014", "internal_reasoning": "0.0000004", "input_cache_read": "0.00000001", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/eu", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.93330512060939, "uptime_last_5m": 99.94105097686953, "uptime_last_1d": 99.87478812348488, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash-lite@google", "name": "Google: Gemini 2.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 1e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000001" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash-lite", "canonicalSlug": "google/gemini-2.5-flash-lite", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...", "uptimeLast30m": 99.9077633268753, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-flash-lite", "model_id": "google/gemini-2.5-flash-lite", "model_name": "Google: Gemini 2.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.0000001", "completion": "0.0000004", "image": "0.0000001", "audio": "0.0000003", "input_audio_cache": "0.00000003", "web_search": "0.014", "internal_reasoning": "0.0000004", "input_cache_read": "0.00000001", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.9077633268753, "uptime_last_5m": 99.92073190870958, "uptime_last_1d": 99.74091942811928, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash-lite@google-ai-studio", "name": "Google: Gemini 2.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 1e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000001" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash-lite", "canonicalSlug": "google/gemini-2.5-flash-lite", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...", "uptimeLast30m": 99.45100920426948, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-flash-lite", "model_id": "google/gemini-2.5-flash-lite", "model_name": "Google: Gemini 2.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.0000001", "completion": "0.0000004", "image": "0.0000001", "audio": "0.0000003", "input_audio_cache": "0.00000003", "web_search": "0.014", "internal_reasoning": "0.0000004", "input_cache_read": "0.00000001", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.45100920426948, "uptime_last_5m": 99.36217893466643, "uptime_last_1d": 99.41044102443999, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash-lite@google-ai-studio", "name": "Google: Gemini 2.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000005" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 5e-8, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000005" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash-lite", "canonicalSlug": "google/gemini-2.5-flash-lite", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...", "uptimeLast30m": 99.45100920426948, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-flash-lite", "model_id": "google/gemini-2.5-flash-lite", "model_name": "Google: Gemini 2.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000005", "completion": "0.0000002", "image": "0.00000005", "audio": "0.00000015", "input_audio_cache": "0.000000015", "web_search": "0.014", "internal_reasoning": "0.0000002", "input_cache_read": "0.000000005", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.45100920426948, "uptime_last_5m": 99.36217893466643, "uptime_last_1d": 99.41044102443999, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash-lite@google-ai-studio", "name": "Google: Gemini 2.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000072" } ], "cacheRead": [ { "amount": 0.018, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000018" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 1.8e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000018" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000072" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash-lite", "canonicalSlug": "google/gemini-2.5-flash-lite", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...", "uptimeLast30m": 99.45100920426948, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-flash-lite", "model_id": "google/gemini-2.5-flash-lite", "model_name": "Google: Gemini 2.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000018", "completion": "0.00000072", "image": "0.00000018", "audio": "0.00000054", "input_audio_cache": "0.000000054", "web_search": "0.014", "internal_reasoning": "0.00000072", "input_cache_read": "0.000000018", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.45100920426948, "uptime_last_5m": 99.36217893466643, "uptime_last_1d": 99.41044102443999, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash-lite:batch@google", "name": "Google: Gemini 2.5 Flash Lite", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000001" } ], "cacheWrite": [], "other": [ { "amount": 5e-8, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000005" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash-lite:batch", "canonicalSlug": "google/gemini-2.5-flash-lite", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-flash-lite:batch", "model_id": "google/gemini-2.5-flash-lite:batch", "model_name": "Google: Gemini 2.5 Flash Lite", "context_length": 1048576, "pricing": { "prompt": "0.00000005", "completion": "0.0000002", "image": "0.00000005", "audio": "0.00000015", "input_audio_cache": "0.00000003", "web_search": "0.014", "internal_reasoning": "0.0000002", "input_cache_read": "0.00000001", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/eu", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-2507@deepinfra", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "output": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-07-25", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...", "uptimeLast30m": 99.59148196436331, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-235b-a22b-07-25", "model_id": "qwen/qwen3-235b-a22b-2507", "model_name": "Qwen: Qwen3 235B A22B Instruct 2507", "context_length": 262144, "pricing": { "prompt": "0.00000009", "completion": "0.00000055", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.59148196436331, "uptime_last_5m": 99.62640099626401, "uptime_last_1d": 96.8194577207277, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-2507@novita", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "output": [ { "amount": 0.58, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000058" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-07-25", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...", "uptimeLast30m": 99.94665244065084, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen3-235b-a22b-07-25", "model_id": "qwen/qwen3-235b-a22b-2507", "model_name": "Qwen: Qwen3 235B A22B Instruct 2507", "context_length": 131072, "pricing": { "prompt": "0.00000009", "completion": "0.00000058", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.94665244065084, "uptime_last_5m": 100, "uptime_last_1d": 98.14331166136395, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-2507@parasail", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-07-25", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...", "uptimeLast30m": 99.96680497925311, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | qwen/qwen3-235b-a22b-07-25", "model_id": "qwen/qwen3-235b-a22b-2507", "model_name": "Qwen: Qwen3 235B A22B Instruct 2507", "context_length": 131072, "pricing": { "prompt": "0.00000014", "completion": "0.0000008", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.96680497925311, "uptime_last_5m": 99.84962406015038, "uptime_last_1d": 99.63180899244418, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-2507@alibaba", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14950000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001495" } ], "output": [ { "amount": 0.5980000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000598" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-07-25", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-235b-a22b-07-25", "model_id": "qwen/qwen3-235b-a22b-2507", "model_name": "Qwen: Qwen3 235B A22B Instruct 2507", "context_length": 131072, "pricing": { "prompt": "0.0000001495", "completion": "0.000000598", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 129024, "supported_parameters": [ "response_format", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "tools", "tool_choice", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96647086919526, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-2507@venice", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-07-25", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...", "uptimeLast30m": 99.16666666666667, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | qwen/qwen3-235b-a22b-07-25", "model_id": "qwen/qwen3-235b-a22b-2507", "model_name": "Qwen: Qwen3 235B A22B Instruct 2507", "context_length": 128000, "pricing": { "prompt": "0.00000015", "completion": "0.00000075", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.16666666666667, "uptime_last_5m": 99.1869918699187, "uptime_last_1d": 95.10882334095132, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-2507@nebius", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-07-25", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 262144, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...", "uptimeLast30m": 98.90234670704012, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | qwen/qwen3-235b-a22b-07-25", "model_id": "qwen/qwen3-235b-a22b-2507", "model_name": "Qwen: Qwen3 235B A22B Instruct 2507", "context_length": 262144, "pricing": { "prompt": "0.0000002", "completion": "0.0000006", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 98.90234670704012, "uptime_last_5m": 100, "uptime_last_1d": 92.1155440685436, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-2507@atlascloud", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000088" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-07-25", "servingProvider": "AtlasCloud", "servingProviderSlug": "atlascloud", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AtlasCloud | qwen/qwen3-235b-a22b-07-25", "model_id": "qwen/qwen3-235b-a22b-2507", "model_name": "Qwen: Qwen3 235B A22B Instruct 2507", "context_length": 131072, "pricing": { "prompt": "0.0000002", "completion": "0.00000088", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "AtlasCloud", "tag": "atlas-cloud/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "min_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "stop", "seed", "logit_bias", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.78775234557777, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-2507@streamlake", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000021" } ], "output": [ { "amount": 0.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000084" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-07-25", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 128000, "maxCompletionTokens": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | qwen/qwen3-235b-a22b-07-25", "model_id": "qwen/qwen3-235b-a22b-2507", "model_name": "Qwen: Qwen3 235B A22B Instruct 2507", "context_length": 128000, "pricing": { "prompt": "0.00000021", "completion": "0.00000084", "discount": 0.4 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "tools", "tool_choice", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.81793076481298, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-2507@google", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000088" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-07-25", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | qwen/qwen3-235b-a22b-07-25", "model_id": "qwen/qwen3-235b-a22b-2507", "model_name": "Qwen: Qwen3 235B A22B Instruct 2507", "context_length": 262144, "pricing": { "prompt": "0.00000022", "completion": "0.00000088", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us-south1", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.8657961552412, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b-2507@google", "name": "Qwen: Qwen3 235B A22B Instruct 2507", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b-2507", "canonicalSlug": "qwen/qwen3-235b-a22b-07-25", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 262144, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": null }, "description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...", "uptimeLast30m": 99.24433249370277, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | qwen/qwen3-235b-a22b-07-25", "model_id": "qwen/qwen3-235b-a22b-2507", "model_name": "Qwen: Qwen3 235B A22B Instruct 2507", "context_length": 262144, "pricing": { "prompt": "0.00000025", "completion": "0.000001", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us-south1", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": 99.24433249370277, "uptime_last_5m": 100, "uptime_last_1d": 98.963169837668, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "moonshotai/kimi-k2@novita", "name": "MoonshotAI: Kimi K2 0711", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5700000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000057" } ], "output": [ { "amount": 2.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000023" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "moonshotai/kimi-k2", "canonicalSlug": "moonshotai/kimi-k2", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 100352, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | moonshotai/kimi-k2", "model_id": "moonshotai/kimi-k2", "model_name": "MoonshotAI: Kimi K2 0711", "context_length": 131072, "pricing": { "prompt": "0.00000057", "completion": "0.0000023", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 100352, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.97574893009985, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "cognitivecomputations/dolphin-mistral-24b-venice-edition@venice", "name": "Venice: Uncensored", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "cognitivecomputations/dolphin-mistral-24b-venice-edition", "canonicalSlug": "venice/uncensored", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 128000, "maxCompletionTokens": 8192, "quantization": "fp16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | venice/uncensored", "model_id": "cognitivecomputations/dolphin-mistral-24b-venice-edition", "model_name": "Venice: Uncensored", "context_length": 128000, "pricing": { "prompt": "0.0000002", "completion": "0.0000009", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp16", "quantization": "fp16", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "tencent/hunyuan-a13b-instruct@siliconflow", "name": "Tencent: Hunyuan A13B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.5700000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000057" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "tencent/hunyuan-a13b-instruct", "canonicalSlug": "tencent/hunyuan-a13b-instruct", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | tencent/hunyuan-a13b-instruct", "model_id": "tencent/hunyuan-a13b-instruct", "model_name": "Tencent: Hunyuan A13B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000014", "completion": "0.00000057", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "max_tokens" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "morph/morph-v3-large@morph", "name": "Morph: Morph V3 Large", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "output": [ { "amount": 1.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000019" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "morph/morph-v3-large", "canonicalSlug": "morph/morph-v3-large", "servingProvider": "Morph", "servingProviderSlug": "morph", "contextLength": 262144, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: {instruction} {initial_code}...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Morph | morph/morph-v3-large", "model_id": "morph/morph-v3-large", "model_name": "Morph: Morph V3 Large", "context_length": 262144, "pricing": { "prompt": "0.0000009", "completion": "0.0000019", "discount": 0 }, "provider_name": "Morph", "tag": "morph/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "morph/morph-v3-fast@morph", "name": "Morph: Morph V3 Fast", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "morph/morph-v3-fast", "canonicalSlug": "morph/morph-v3-fast", "servingProvider": "Morph", "servingProviderSlug": "morph", "contextLength": 81920, "maxCompletionTokens": 38000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: {instruction} {initial_code} {edit_snippet}...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Morph | morph/morph-v3-fast", "model_id": "morph/morph-v3-fast", "model_name": "Morph: Morph V3 Fast", "context_length": 81920, "pricing": { "prompt": "0.0000008", "completion": "0.0000012", "discount": 0 }, "provider_name": "Morph", "tag": "morph", "quantization": "unknown", "max_completion_tokens": 38000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "baidu/ernie-4.5-vl-424b-a47b@novita", "name": "Baidu: ERNIE 4.5 VL 424B A47B ", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000042" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "baidu/ernie-4.5-vl-424b-a47b", "canonicalSlug": "baidu/ernie-4.5-vl-424b-a47b", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 123000, "maxCompletionTokens": 16000, "quantization": "fp16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | baidu/ernie-4.5-vl-424b-a47b", "model_id": "baidu/ernie-4.5-vl-424b-a47b", "model_name": "Baidu: ERNIE 4.5 VL 424B A47B ", "context_length": 123000, "pricing": { "prompt": "0.00000042", "completion": "0.00000125", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp16", "quantization": "fp16", "max_completion_tokens": 16000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-small-3.2-24b-instruct@deepinfra", "name": "Mistral: Mistral Small 3.2 24B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-small-3.2-24b-instruct", "canonicalSlug": "mistralai/mistral-small-3.2-24b-instruct-2506", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...", "uptimeLast30m": 99.88424251193749, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | mistralai/mistral-small-3.2-24b-instruct-2506", "model_id": "mistralai/mistral-small-3.2-24b-instruct", "model_name": "Mistral: Mistral Small 3.2 24B", "context_length": 128000, "pricing": { "prompt": "0.000000075", "completion": "0.0000002", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.88424251193749, "uptime_last_5m": 100, "uptime_last_1d": 99.64743012282479, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-small-3.2-24b-instruct@parasail", "name": "Mistral: Mistral Small 3.2 24B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-small-3.2-24b-instruct", "canonicalSlug": "mistralai/mistral-small-3.2-24b-instruct-2506", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...", "uptimeLast30m": 99.68173138128581, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | mistralai/mistral-small-3.2-24b-instruct-2506", "model_id": "mistralai/mistral-small-3.2-24b-instruct", "model_name": "Mistral: Mistral Small 3.2 24B", "context_length": 131072, "pricing": { "prompt": "0.00000009", "completion": "0.0000003", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.68173138128581, "uptime_last_5m": 99.46236559139786, "uptime_last_1d": 99.172291728123, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-small-3.2-24b-instruct@venice", "name": "Mistral: Mistral Small 3.2 24B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000009375" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-small-3.2-24b-instruct", "canonicalSlug": "mistralai/mistral-small-3.2-24b-instruct-2506", "servingProvider": "Venice", "servingProviderSlug": "venice", "contextLength": 256000, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...", "uptimeLast30m": 99.845449681983, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Venice | mistralai/mistral-small-3.2-24b-instruct-2506", "model_id": "mistralai/mistral-small-3.2-24b-instruct", "model_name": "Mistral: Mistral Small 3.2 24B", "context_length": 256000, "pricing": { "prompt": "0.00000009375", "completion": "0.00000025", "discount": 0 }, "provider_name": "Venice", "tag": "venice/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.845449681983, "uptime_last_5m": 100, "uptime_last_1d": 99.54419308483457, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m1@minimax", "name": "MiniMax: MiniMax M1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m1", "canonicalSlug": "minimax/minimax-m1", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 1000000, "maxCompletionTokens": 40000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/minimax-m1", "model_id": "minimax/minimax-m1", "model_name": "MiniMax: MiniMax M1", "context_length": 1000000, "pricing": { "prompt": "0.0000004", "completion": "0.0000022", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax", "quantization": "unknown", "max_completion_tokens": 40000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-m1@novita", "name": "MiniMax: MiniMax M1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-m1", "canonicalSlug": "minimax/minimax-m1", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 1000000, "maxCompletionTokens": 40000, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | minimax/minimax-m1", "model_id": "minimax/minimax-m1", "model_name": "MiniMax: MiniMax M1", "context_length": 1000000, "pricing": { "prompt": "0.00000055", "completion": "0.0000022", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 40000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.89594172736732, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash@google", "name": "Google: Gemini 2.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 3e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000003" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash", "canonicalSlug": "google/gemini-2.5-flash", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "file", "image", "text", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...", "uptimeLast30m": 85.55304740406321, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-flash", "model_id": "google/gemini-2.5-flash", "model_name": "Google: Gemini 2.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "image": "0.0000003", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.0000025", "input_cache_read": "0.00000003", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/eu", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": -2, "uptime_last_30m": 85.55304740406321, "uptime_last_5m": 57.391304347826086, "uptime_last_1d": 90.40769697962962, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash@google", "name": "Google: Gemini 2.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 3e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000003" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash", "canonicalSlug": "google/gemini-2.5-flash", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "file", "image", "text", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...", "uptimeLast30m": 99.85846367300722, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-flash", "model_id": "google/gemini-2.5-flash", "model_name": "Google: Gemini 2.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "image": "0.0000003", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.0000025", "input_cache_read": "0.00000003", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.85846367300722, "uptime_last_5m": 99.82649885709564, "uptime_last_1d": 99.02496570723434, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash@google", "name": "Google: Gemini 2.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000054" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.054, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000054" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 5.4e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000054" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash", "canonicalSlug": "google/gemini-2.5-flash", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "file", "image", "text", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...", "uptimeLast30m": 99.85846367300722, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-flash", "model_id": "google/gemini-2.5-flash", "model_name": "Google: Gemini 2.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000054", "completion": "0.0000045", "image": "0.00000054", "audio": "0.0000018", "input_audio_cache": "0.00000018", "web_search": "0.014", "internal_reasoning": "0.0000045", "input_cache_read": "0.000000054", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global/priority", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.85846367300722, "uptime_last_5m": 99.82649885709564, "uptime_last_1d": 99.02496570723434, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash@google-ai-studio", "name": "Google: Gemini 2.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 3e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000003" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash", "canonicalSlug": "google/gemini-2.5-flash", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "file", "image", "text", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...", "uptimeLast30m": 99.93325917686317, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-flash", "model_id": "google/gemini-2.5-flash", "model_name": "Google: Gemini 2.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "image": "0.0000003", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.0000025", "input_cache_read": "0.00000003", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.93325917686317, "uptime_last_5m": 99.91023339317773, "uptime_last_1d": 99.930891902446, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash@google-ai-studio", "name": "Google: Gemini 2.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000015" } ], "cacheWrite": [ { "amount": 0.0416666666666667, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000416666666666667" } ], "other": [ { "amount": 1.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000015" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash", "canonicalSlug": "google/gemini-2.5-flash", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "file", "image", "text", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...", "uptimeLast30m": 99.93325917686317, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-flash", "model_id": "google/gemini-2.5-flash", "model_name": "Google: Gemini 2.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000015", "completion": "0.00000125", "image": "0.00000015", "audio": "0.0000005", "input_audio_cache": "0.00000005", "web_search": "0.014", "internal_reasoning": "0.00000125", "input_cache_read": "0.000000015", "input_cache_write": "0.0000000416666666666667", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.93325917686317, "uptime_last_5m": 99.91023339317773, "uptime_last_1d": 99.930891902446, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash@google-ai-studio", "name": "Google: Gemini 2.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000054" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.054, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000054" } ], "cacheWrite": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "other": [ { "amount": 5.4e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000054" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash", "canonicalSlug": "google/gemini-2.5-flash", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "file", "image", "text", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...", "uptimeLast30m": 99.93325917686317, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-flash", "model_id": "google/gemini-2.5-flash", "model_name": "Google: Gemini 2.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000054", "completion": "0.0000045", "image": "0.00000054", "audio": "0.0000018", "input_audio_cache": "0.00000018", "web_search": "0.014", "internal_reasoning": "0.0000045", "input_cache_read": "0.000000054", "input_cache_write": "0.00000015", "discount": 0 }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.93325917686317, "uptime_last_5m": 99.91023339317773, "uptime_last_1d": 99.930891902446, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash@google", "name": "Google: Gemini 2.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [ { "amount": 0.0833333333333333, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000833333333333333" } ], "other": [ { "amount": 3e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0000003" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash", "canonicalSlug": "google/gemini-2.5-flash", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "file", "image", "text", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...", "uptimeLast30m": 88.76825159116436, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-flash", "model_id": "google/gemini-2.5-flash", "model_name": "Google: Gemini 2.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.0000003", "completion": "0.0000025", "image": "0.0000003", "audio": "0.000001", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.0000025", "input_cache_read": "0.00000003", "input_cache_write": "0.0000000833333333333333", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": -2, "uptime_last_30m": 88.76825159116436, "uptime_last_5m": 85.02673796791443, "uptime_last_1d": 88.12884091654428, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-flash:batch@google", "name": "Google: Gemini 2.5 Flash", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [ { "amount": 1.5e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000015" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-flash:batch", "canonicalSlug": "google/gemini-2.5-flash", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65535, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "file", "image", "text", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-flash:batch", "model_id": "google/gemini-2.5-flash:batch", "model_name": "Google: Gemini 2.5 Flash", "context_length": 1048576, "pricing": { "prompt": "0.00000015", "completion": "0.00000125", "image": "0.00000015", "audio": "0.0000005", "input_audio_cache": "0.0000001", "web_search": "0.014", "internal_reasoning": "0.00000125", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65535, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro", "canonicalSlug": "google/gemini-2.5-pro", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 99.74883837749591, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.74883837749591, "uptime_last_5m": 100, "uptime_last_1d": 99.55219292977114, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "other": [ { "amount": 6.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000625" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro", "canonicalSlug": "google/gemini-2.5-pro", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 99.74883837749591, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "image": "0.000000625", "audio": "0.000000625", "input_audio_cache": "0.0000000625", "web_search": "0.014", "internal_reasoning": "0.000005", "input_cache_read": "0.0000000625", "input_cache_write": "0.0000001875", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.00000125", "completion": "0.0000075", "audio": "0.00000125", "input_audio_cache": "0.000000125", "input_cache_read": "0.000000125" } ] }, "provider_name": "Google", "tag": "google-vertex/global/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.74883837749591, "uptime_last_5m": 100, "uptime_last_1d": 99.55219292977114, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.22499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000225" } ], "cacheWrite": [ { "amount": 0.675, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000675" } ], "other": [ { "amount": 0.00000225, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000225" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro", "canonicalSlug": "google/gemini-2.5-pro", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 99.74883837749591, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000225", "completion": "0.000018", "image": "0.00000225", "audio": "0.00000225", "input_audio_cache": "0.000000225", "web_search": "0.014", "internal_reasoning": "0.000018", "input_cache_read": "0.000000225", "input_cache_write": "0.000000675", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000045", "completion": "0.000027", "audio": "0.0000045", "input_audio_cache": "0.00000045", "input_cache_read": "0.00000045" } ] }, "provider_name": "Google", "tag": "google-vertex/global/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.74883837749591, "uptime_last_5m": 100, "uptime_last_1d": 99.55219292977114, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro", "canonicalSlug": "google/gemini-2.5-pro", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google", "tag": "google-vertex/eu", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 74.03366583541147, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro", "canonicalSlug": "google/gemini-2.5-pro", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 35.58246828143022, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro@google-ai-studio", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro", "canonicalSlug": "google/gemini-2.5-pro", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 94.0625, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "status": -2, "uptime_last_30m": 94.0625, "uptime_last_5m": null, "uptime_last_1d": 95.01701289398281, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro@google-ai-studio", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "other": [ { "amount": 6.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000625" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro", "canonicalSlug": "google/gemini-2.5-pro", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 94.0625, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "image": "0.000000625", "audio": "0.000000625", "input_audio_cache": "0.0000000625", "web_search": "0.014", "internal_reasoning": "0.000005", "input_cache_read": "0.0000000625", "input_cache_write": "0.0000001875", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.00000125", "completion": "0.0000075", "audio": "0.00000125", "input_audio_cache": "0.000000125", "input_cache_read": "0.000000125" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "status": -2, "uptime_last_30m": 94.0625, "uptime_last_5m": null, "uptime_last_1d": 95.01701289398281, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro@google-ai-studio", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.22499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000225" } ], "cacheWrite": [ { "amount": 0.675, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000675" } ], "other": [ { "amount": 0.00000225, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000225" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro", "canonicalSlug": "google/gemini-2.5-pro", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 94.0625, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000225", "completion": "0.000018", "image": "0.00000225", "audio": "0.00000225", "input_audio_cache": "0.000000225", "web_search": "0.014", "internal_reasoning": "0.000018", "input_cache_read": "0.000000225", "input_cache_write": "0.000000675", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000045", "completion": "0.000027", "audio": "0.0000045", "input_audio_cache": "0.00000045", "input_cache_read": "0.00000045" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "status": -2, "uptime_last_30m": 94.0625, "uptime_last_5m": null, "uptime_last_1d": 95.01701289398281, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro:batch@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [], "other": [ { "amount": 6.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000625" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro:batch", "canonicalSlug": "google/gemini-2.5-pro", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro:batch", "model_id": "google/gemini-2.5-pro:batch", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "image": "0.000000625", "audio": "0.000000625", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.000005", "input_cache_read": "0.000000125", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.00000125", "completion": "0.0000075", "audio": "0.00000125", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o3-pro@openai", "name": "OpenAI: o3 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "output": [ { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00008" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o3-pro", "canonicalSlug": "openai/o3-pro-2025-06-10", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "file", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o3-pro-2025-06-10", "model_id": "openai/o3-pro", "model_name": "OpenAI: o3 Pro", "context_length": 200000, "pricing": { "prompt": "0.00002", "completion": "0.00008", "web_search": "0.01", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o3-pro:batch@openai", "name": "OpenAI: o3 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00004" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o3-pro:batch", "canonicalSlug": "openai/o3-pro-2025-06-10", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "file", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o3-pro-2025-06-10:batch", "model_id": "openai/o3-pro:batch", "model_name": "OpenAI: o3 Pro", "context_length": 200000, "pricing": { "prompt": "0.00001", "completion": "0.00004", "web_search": "0.01", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview", "canonicalSlug": "google/gemini-2.5-pro-preview-06-05", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio->text", "input_modalities": [ "file", "image", "text", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 99.74883837749591, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.74883837749591, "uptime_last_5m": 100, "uptime_last_1d": 99.55219292977114, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "other": [ { "amount": 6.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000625" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview", "canonicalSlug": "google/gemini-2.5-pro-preview-06-05", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio->text", "input_modalities": [ "file", "image", "text", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 99.74883837749591, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "image": "0.000000625", "audio": "0.000000625", "input_audio_cache": "0.0000000625", "web_search": "0.014", "internal_reasoning": "0.000005", "input_cache_read": "0.0000000625", "input_cache_write": "0.0000001875", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.00000125", "completion": "0.0000075", "audio": "0.00000125", "input_audio_cache": "0.000000125", "input_cache_read": "0.000000125" } ] }, "provider_name": "Google", "tag": "google-vertex/global/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.74883837749591, "uptime_last_5m": 100, "uptime_last_1d": 99.55219292977114, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.22499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000225" } ], "cacheWrite": [ { "amount": 0.675, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000675" } ], "other": [ { "amount": 0.00000225, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000225" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview", "canonicalSlug": "google/gemini-2.5-pro-preview-06-05", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio->text", "input_modalities": [ "file", "image", "text", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 99.74883837749591, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000225", "completion": "0.000018", "image": "0.00000225", "audio": "0.00000225", "input_audio_cache": "0.000000225", "web_search": "0.014", "internal_reasoning": "0.000018", "input_cache_read": "0.000000225", "input_cache_write": "0.000000675", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000045", "completion": "0.000027", "audio": "0.0000045", "input_audio_cache": "0.00000045", "input_cache_read": "0.00000045" } ] }, "provider_name": "Google", "tag": "google-vertex/global/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.74883837749591, "uptime_last_5m": 100, "uptime_last_1d": 99.55219292977114, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview", "canonicalSlug": "google/gemini-2.5-pro-preview-06-05", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio->text", "input_modalities": [ "file", "image", "text", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google", "tag": "google-vertex/eu", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 74.03366583541147, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview", "canonicalSlug": "google/gemini-2.5-pro-preview-06-05", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio->text", "input_modalities": [ "file", "image", "text", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 35.58246828143022, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview@google-ai-studio", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview", "canonicalSlug": "google/gemini-2.5-pro-preview-06-05", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file+audio->text", "input_modalities": [ "file", "image", "text", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 94.0625, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "status": -2, "uptime_last_30m": 94.0625, "uptime_last_5m": null, "uptime_last_1d": 95.01701289398281, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview@google-ai-studio", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "other": [ { "amount": 6.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000625" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview", "canonicalSlug": "google/gemini-2.5-pro-preview-06-05", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file+audio->text", "input_modalities": [ "file", "image", "text", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 94.0625, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "image": "0.000000625", "audio": "0.000000625", "input_audio_cache": "0.0000000625", "web_search": "0.014", "internal_reasoning": "0.000005", "input_cache_read": "0.0000000625", "input_cache_write": "0.0000001875", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.00000125", "completion": "0.0000075", "audio": "0.00000125", "input_audio_cache": "0.000000125", "input_cache_read": "0.000000125" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "status": -2, "uptime_last_30m": 94.0625, "uptime_last_5m": null, "uptime_last_1d": 95.01701289398281, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview@google-ai-studio", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.22499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000225" } ], "cacheWrite": [ { "amount": 0.675, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000675" } ], "other": [ { "amount": 0.00000225, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000225" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview", "canonicalSlug": "google/gemini-2.5-pro-preview-06-05", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file+audio->text", "input_modalities": [ "file", "image", "text", "audio" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 94.0625, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000225", "completion": "0.000018", "image": "0.00000225", "audio": "0.00000225", "input_audio_cache": "0.000000225", "web_search": "0.014", "internal_reasoning": "0.000018", "input_cache_read": "0.000000225", "input_cache_write": "0.000000675", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000045", "completion": "0.000027", "audio": "0.0000045", "input_audio_cache": "0.00000045", "input_cache_read": "0.00000045" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "status": -2, "uptime_last_30m": 94.0625, "uptime_last_5m": null, "uptime_last_1d": 95.01701289398281, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-r1-0528@deepinfra", "name": "DeepSeek: R1 0528", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 2.1500000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000215" } ], "cacheRead": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-r1-0528", "canonicalSlug": "deepseek/deepseek-r1-0528", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 163840, "maxCompletionTokens": 32768, "quantization": "fp4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-r1" }, "description": "May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | deepseek/deepseek-r1-0528", "model_id": "deepseek/deepseek-r1-0528", "model_name": "DeepSeek: R1 0528", "context_length": 163840, "pricing": { "prompt": "0.0000005", "completion": "0.00000215", "input_cache_read": "0.00000035", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.87992663649439, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-r1-0528@siliconflow", "name": "DeepSeek: R1 0528", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 2.1799999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000218" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-r1-0528", "canonicalSlug": "deepseek/deepseek-r1-0528", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 163840, "maxCompletionTokens": 163840, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-r1" }, "description": "May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...", "uptimeLast30m": 99.2843201040989, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | deepseek/deepseek-r1-0528", "model_id": "deepseek/deepseek-r1-0528", "model_name": "DeepSeek: R1 0528", "context_length": 163840, "pricing": { "prompt": "0.0000005", "completion": "0.00000218", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 163840, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 99.2843201040989, "uptime_last_5m": 100, "uptime_last_1d": 99.43326636480853, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-r1-0528@streamlake", "name": "DeepSeek: R1 0528", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5710000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000571" } ], "output": [ { "amount": 2.2859999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002286" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-r1-0528", "canonicalSlug": "deepseek/deepseek-r1-0528", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 128000, "maxCompletionTokens": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-r1" }, "description": "May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...", "uptimeLast30m": 99.83333333333333, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | deepseek/deepseek-r1-0528", "model_id": "deepseek/deepseek-r1-0528", "model_name": "DeepSeek: R1 0528", "context_length": 128000, "pricing": { "prompt": "0.000000571", "completion": "0.000002286", "discount": 0 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "structured_outputs", "logprobs", "top_logprobs", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "status": 0, "uptime_last_30m": 99.83333333333333, "uptime_last_5m": 100, "uptime_last_1d": 99.77618865658606, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-r1-0528@novita", "name": "DeepSeek: R1 0528", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-r1-0528", "canonicalSlug": "deepseek/deepseek-r1-0528", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 163840, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-r1" }, "description": "May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-r1-0528", "model_id": "deepseek/deepseek-r1-0528", "model_name": "DeepSeek: R1 0528", "context_length": 163840, "pricing": { "prompt": "0.0000007", "completion": "0.0000025", "input_cache_read": "0.00000035", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.41060903732809, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-opus-4@google", "name": "Anthropic: Claude Opus 4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-opus-4", "canonicalSlug": "anthropic/claude-4-opus-20250522", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 200000, "maxCompletionTokens": 32000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4-opus-20250522", "model_id": "anthropic/claude-opus-4", "model_name": "Anthropic: Claude Opus 4", "context_length": 200000, "pricing": { "prompt": "0.000015", "completion": "0.000075", "web_search": "0.01", "input_cache_read": "0.0000015", "input_cache_write": "0.00001875", "input_cache_write_1h": "0.00003", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4@google", "name": "Anthropic: Claude Sonnet 4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4", "canonicalSlug": "anthropic/claude-4-sonnet-20250522", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4-sonnet-20250522", "model_id": "anthropic/claude-sonnet-4", "model_name": "Anthropic: Claude Sonnet 4", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "input_cache_write": "0.0000075", "input_cache_write_1h": "0.000012" } ] }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.98594400089958, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4@amazon-bedrock", "name": "Anthropic: Claude Sonnet 4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4", "canonicalSlug": "anthropic/claude-4-sonnet-20250522", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4-sonnet-20250522", "model_id": "anthropic/claude-sonnet-4", "model_name": "Anthropic: Claude Sonnet 4", "context_length": 200000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "input_cache_write": "0.0000075", "input_cache_write_1h": "0.000012" } ] }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4@amazon-bedrock", "name": "Anthropic: Claude Sonnet 4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4", "canonicalSlug": "anthropic/claude-4-sonnet-20250522", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 200000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...", "uptimeLast30m": 99.91026024528867, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-4-sonnet-20250522", "model_id": "anthropic/claude-sonnet-4", "model_name": "Anthropic: Claude Sonnet 4", "context_length": 200000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "input_cache_write": "0.0000075", "input_cache_write_1h": "0.000012" } ] }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.91026024528867, "uptime_last_5m": 100, "uptime_last_1d": 99.94196898130453, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4@google", "name": "Anthropic: Claude Sonnet 4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4", "canonicalSlug": "anthropic/claude-4-sonnet-20250522", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4-sonnet-20250522", "model_id": "anthropic/claude-sonnet-4", "model_name": "Anthropic: Claude Sonnet 4", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "input_cache_write": "0.0000075", "input_cache_write_1h": "0.000012" } ] }, "provider_name": "Google", "tag": "google-vertex/europe", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-sonnet-4@google", "name": "Anthropic: Claude Sonnet 4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-sonnet-4", "canonicalSlug": "anthropic/claude-4-sonnet-20250522", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1000000, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | anthropic/claude-4-sonnet-20250522", "model_id": "anthropic/claude-sonnet-4", "model_name": "Anthropic: Claude Sonnet 4", "context_length": 1000000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.01", "input_cache_read": "0.0000003", "input_cache_write": "0.00000375", "input_cache_write_1h": "0.000006", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.000006", "completion": "0.0000225", "input_cache_read": "0.0000006", "input_cache_write": "0.0000075", "input_cache_write_1h": "0.000012" } ] }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "top_p", "temperature", "stop", "reasoning", "include_reasoning", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-3n-e4b-it@together", "name": "Google: Gemma 3n 4B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "output": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-3n-e4b-it", "canonicalSlug": "google/gemma-3n-e4b-it", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...", "uptimeLast30m": 99.98144368157358, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | google/gemma-3n-e4b-it", "model_id": "google/gemma-3n-e4b-it", "model_name": "Google: Gemma 3n 4B", "context_length": 32768, "pricing": { "prompt": "0.00000006", "completion": "0.00000012", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.98144368157358, "uptime_last_5m": 100, "uptime_last_1d": 99.88048663793359, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-medium-3@mistral", "name": "Mistral: Mistral Medium 3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-medium-3", "canonicalSlug": "mistralai/mistral-medium-3", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/mistral-medium-3", "model_id": "mistralai/mistral-medium-3", "model_name": "Mistral: Mistral Medium 3", "context_length": 131072, "pricing": { "prompt": "0.0000004", "completion": "0.000002", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.91980753809142, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview-05-06@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview-05-06", "canonicalSlug": "google/gemini-2.5-pro-preview-03-25", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 99.74883837749591, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google", "tag": "google-vertex/global", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.74883837749591, "uptime_last_5m": 100, "uptime_last_1d": 99.55219292977114, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview-05-06@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "other": [ { "amount": 6.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000625" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview-05-06", "canonicalSlug": "google/gemini-2.5-pro-preview-03-25", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 99.74883837749591, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "image": "0.000000625", "audio": "0.000000625", "input_audio_cache": "0.0000000625", "web_search": "0.014", "internal_reasoning": "0.000005", "input_cache_read": "0.0000000625", "input_cache_write": "0.0000001875", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.00000125", "completion": "0.0000075", "audio": "0.00000125", "input_audio_cache": "0.000000125", "input_cache_read": "0.000000125" } ] }, "provider_name": "Google", "tag": "google-vertex/global/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.74883837749591, "uptime_last_5m": 100, "uptime_last_1d": 99.55219292977114, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview-05-06@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.22499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000225" } ], "cacheWrite": [ { "amount": 0.675, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000675" } ], "other": [ { "amount": 0.00000225, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000225" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview-05-06", "canonicalSlug": "google/gemini-2.5-pro-preview-03-25", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 99.74883837749591, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000225", "completion": "0.000018", "image": "0.00000225", "audio": "0.00000225", "input_audio_cache": "0.000000225", "web_search": "0.014", "internal_reasoning": "0.000018", "input_cache_read": "0.000000225", "input_cache_write": "0.000000675", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000045", "completion": "0.000027", "audio": "0.0000045", "input_audio_cache": "0.00000045", "input_cache_read": "0.00000045" } ] }, "provider_name": "Google", "tag": "google-vertex/global/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": 99.74883837749591, "uptime_last_5m": 100, "uptime_last_1d": 99.55219292977114, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview-05-06@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview-05-06", "canonicalSlug": "google/gemini-2.5-pro-preview-03-25", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google", "tag": "google-vertex/eu", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 74.03366583541147, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview-05-06@google", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview-05-06", "canonicalSlug": "google/gemini-2.5-pro-preview-03-25", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google", "tag": "google-vertex/us", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice", "stop" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 35.58246828143022, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview-05-06@google-ai-studio", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000125" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000375" } ], "other": [ { "amount": 0.00000125, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000125" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview-05-06", "canonicalSlug": "google/gemini-2.5-pro-preview-03-25", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 94.0625, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000125", "completion": "0.00001", "image": "0.00000125", "audio": "0.00000125", "input_audio_cache": "0.000000125", "web_search": "0.014", "internal_reasoning": "0.00001", "input_cache_read": "0.000000125", "input_cache_write": "0.000000375", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000025", "completion": "0.000015", "audio": "0.0000025", "input_audio_cache": "0.00000025", "input_cache_read": "0.00000025" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "status": -2, "uptime_last_30m": 94.0625, "uptime_last_5m": null, "uptime_last_1d": 95.01701289398281, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview-05-06@google-ai-studio", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.0625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000625" } ], "cacheWrite": [ { "amount": 0.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001875" } ], "other": [ { "amount": 6.25e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000000625" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview-05-06", "canonicalSlug": "google/gemini-2.5-pro-preview-03-25", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 94.0625, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.000000625", "completion": "0.000005", "image": "0.000000625", "audio": "0.000000625", "input_audio_cache": "0.0000000625", "web_search": "0.014", "internal_reasoning": "0.000005", "input_cache_read": "0.0000000625", "input_cache_write": "0.0000001875", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.00000125", "completion": "0.0000075", "audio": "0.00000125", "input_audio_cache": "0.000000125", "input_cache_read": "0.000000125" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/flex", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "status": -2, "uptime_last_30m": 94.0625, "uptime_last_5m": null, "uptime_last_1d": 95.01701289398281, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemini-2.5-pro-preview-05-06@google-ai-studio", "name": "Google: Gemini 2.5 Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.22499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000225" } ], "cacheWrite": [ { "amount": 0.675, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000675" } ], "other": [ { "amount": 0.00000225, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00000225" }, { "amount": 0.014, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.014" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ] } ], "metadata": { "source": "openrouter", "modelId": "google/gemini-2.5-pro-preview-05-06", "canonicalSlug": "google/gemini-2.5-pro-preview-03-25", "servingProvider": "Google AI Studio", "servingProviderSlug": "google-ai-studio", "contextLength": 1048576, "maxCompletionTokens": 65536, "quantization": "unknown", "status": -2, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file+audio+video->text", "input_modalities": [ "text", "image", "file", "audio", "video" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": null }, "description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...", "uptimeLast30m": 94.0625, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google AI Studio | google/gemini-2.5-pro", "model_id": "google/gemini-2.5-pro", "model_name": "Google: Gemini 2.5 Pro", "context_length": 1048576, "pricing": { "prompt": "0.00000225", "completion": "0.000018", "image": "0.00000225", "audio": "0.00000225", "input_audio_cache": "0.000000225", "web_search": "0.014", "internal_reasoning": "0.000018", "input_cache_read": "0.000000225", "input_cache_write": "0.000000675", "discount": 0, "overrides": [ { "min_prompt_tokens": 200000, "prompt": "0.0000045", "completion": "0.000027", "audio": "0.0000045", "input_audio_cache": "0.00000045", "input_cache_read": "0.00000045" } ] }, "provider_name": "Google AI Studio", "tag": "google-ai-studio/priority", "quantization": "unknown", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "tools", "tool_choice" ], "status": -2, "uptime_last_30m": 94.0625, "uptime_last_5m": null, "uptime_last_1d": 95.01701289398281, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "arcee-ai/virtuoso-large@together", "name": "Arcee AI: Virtuoso Large", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "arcee-ai/virtuoso-large", "canonicalSlug": "arcee-ai/virtuoso-large", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 131072, "maxCompletionTokens": 64000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | arcee-ai/virtuoso-large", "model_id": "arcee-ai/virtuoso-large", "model_name": "Arcee AI: Virtuoso Large", "context_length": 131072, "pricing": { "prompt": "0.00000075", "completion": "0.0000012", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": 64000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-guard-4-12b@deepinfra", "name": "Meta: Llama Guard 4 12B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-guard-4-12b", "canonicalSlug": "meta-llama/llama-guard-4-12b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 163840, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...", "uptimeLast30m": 99.91210079109288, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | meta-llama/llama-guard-4-12b", "model_id": "meta-llama/llama-guard-4-12b", "model_name": "Meta: Llama Guard 4 12B", "context_length": 163840, "pricing": { "prompt": "0.00000018", "completion": "0.00000018", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "status": 0, "uptime_last_30m": 99.91210079109288, "uptime_last_5m": 99.54337899543378, "uptime_last_1d": 99.44448938866867, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-guard-4-12b@together", "name": "Meta: Llama Guard 4 12B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-guard-4-12b", "canonicalSlug": "meta-llama/llama-guard-4-12b", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 1048576, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "image", "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...", "uptimeLast30m": 99.95083579154375, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | meta-llama/llama-guard-4-12b", "model_id": "meta-llama/llama-guard-4-12b", "model_name": "Meta: Llama Guard 4 12B", "context_length": 1048576, "pricing": { "prompt": "0.0000002", "completion": "0.0000002", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p" ], "status": 0, "uptime_last_30m": 99.95083579154375, "uptime_last_5m": 100, "uptime_last_1d": 99.91701638333689, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-30b-a3b@deepinfra", "name": "Qwen: Qwen3 30B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-30b-a3b", "canonicalSlug": "qwen/qwen3-30b-a3b-04-28", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 40960, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "tools", "tool_choice", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...", "uptimeLast30m": 95.24647887323944, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-30b-a3b-04-28", "model_id": "qwen/qwen3-30b-a3b", "model_name": "Qwen: Qwen3 30B A3B", "context_length": 40960, "pricing": { "prompt": "0.00000012", "completion": "0.0000005", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "tools", "tool_choice", "logit_bias" ], "status": 0, "uptime_last_30m": 95.24647887323944, "uptime_last_5m": 100, "uptime_last_1d": 99.38790500285644, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-30b-a3b@alibaba", "name": "Qwen: Qwen3 30B A3B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000052" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-30b-a3b", "canonicalSlug": "qwen/qwen3-30b-a3b-04-28", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-30b-a3b-04-28", "model_id": "qwen/qwen3-30b-a3b", "model_name": "Qwen: Qwen3 30B A3B", "context_length": 131072, "pricing": { "prompt": "0.00000013", "completion": "0.00000052", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": 98304, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tool_choice", "tools", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.97422126745435, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-8b@alibaba", "name": "Qwen: Qwen3 8B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.117, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000117" } ], "output": [ { "amount": 0.45499999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000455" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-8b", "canonicalSlug": "qwen/qwen3-8b-04-28", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-8b-04-28", "model_id": "qwen/qwen3-8b", "model_name": "Qwen: Qwen3 8B", "context_length": 131072, "pricing": { "prompt": "0.000000117", "completion": "0.000000455", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": 98304, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-14b@nextbit", "name": "Qwen: Qwen3 14B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-14b", "canonicalSlug": "qwen/qwen3-14b-04-28", "servingProvider": "NextBit", "servingProviderSlug": "nextbit", "contextLength": 40960, "maxCompletionTokens": 40960, "quantization": "int4", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...", "uptimeLast30m": 99.95814148179154, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "NextBit | qwen/qwen3-14b-04-28", "model_id": "qwen/qwen3-14b", "model_name": "Qwen: Qwen3 14B", "context_length": 40960, "pricing": { "prompt": "0.0000001", "completion": "0.00000022", "discount": 0 }, "provider_name": "NextBit", "tag": "nextbit/int4", "quantization": "int4", "max_completion_tokens": 40960, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.95814148179154, "uptime_last_5m": 100, "uptime_last_1d": 97.6304290996004, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-14b@deepinfra", "name": "Qwen: Qwen3 14B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000012" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-14b", "canonicalSlug": "qwen/qwen3-14b-04-28", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 40960, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...", "uptimeLast30m": 99.87995198079231, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-14b-04-28", "model_id": "qwen/qwen3-14b", "model_name": "Qwen: Qwen3 14B", "context_length": 40960, "pricing": { "prompt": "0.00000012", "completion": "0.00000024", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.87995198079231, "uptime_last_5m": 100, "uptime_last_1d": 98.33057252303551, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-14b@alibaba", "name": "Qwen: Qwen3 14B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22749999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002275" } ], "output": [ { "amount": 0.9099999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000091" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-14b", "canonicalSlug": "qwen/qwen3-14b-04-28", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-14b-04-28", "model_id": "qwen/qwen3-14b", "model_name": "Qwen: Qwen3 14B", "context_length": 131072, "pricing": { "prompt": "0.0000002275", "completion": "0.00000091", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": 98304, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-32b@deepinfra", "name": "Qwen: Qwen3 32B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000028" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-32b", "canonicalSlug": "qwen/qwen3-32b-04-28", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 40960, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...", "uptimeLast30m": 99.99683153258768, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen3-32b-04-28", "model_id": "qwen/qwen3-32b", "model_name": "Qwen: Qwen3 32B", "context_length": 40960, "pricing": { "prompt": "0.00000008", "completion": "0.00000028", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.99683153258768, "uptime_last_5m": 100, "uptime_last_1d": 99.93341284099338, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-32b@nebius", "name": "Qwen: Qwen3 32B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-32b", "canonicalSlug": "qwen/qwen3-32b-04-28", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 40960, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "repetition_penalty" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | qwen/qwen3-32b-04-28", "model_id": "qwen/qwen3-32b", "model_name": "Qwen: Qwen3 32B", "context_length": 40960, "pricing": { "prompt": "0.0000001", "completion": "0.0000003", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/base", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "repetition_penalty" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 97.1901377581795, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-32b@siliconflow", "name": "Qwen: Qwen3 32B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "output": [ { "amount": 0.5700000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000057" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-32b", "canonicalSlug": "qwen/qwen3-32b-04-28", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...", "uptimeLast30m": 99.44639232330688, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | qwen/qwen3-32b-04-28", "model_id": "qwen/qwen3-32b", "model_name": "Qwen: Qwen3 32B", "context_length": 131072, "pricing": { "prompt": "0.00000014", "completion": "0.00000057", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 99.44639232330688, "uptime_last_5m": 100, "uptime_last_1d": 89.61926390686293, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-32b@groq", "name": "Qwen: Qwen3 32B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000029" } ], "output": [ { "amount": 0.59, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000059" } ], "cacheRead": [ { "amount": 0.145, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000145" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-32b", "canonicalSlug": "qwen/qwen3-32b-04-28", "servingProvider": "Groq", "servingProviderSlug": "groq", "contextLength": 131072, "maxCompletionTokens": 40960, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Groq | qwen/qwen3-32b-04-28", "model_id": "qwen/qwen3-32b", "model_name": "Qwen: Qwen3 32B", "context_length": 131072, "pricing": { "prompt": "0.00000029", "completion": "0.00000059", "input_cache_read": "0.000000145", "discount": 0 }, "provider_name": "Groq", "tag": "groq", "quantization": "unknown", "max_completion_tokens": 40960, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9909843191551, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen3-235b-a22b@alibaba", "name": "Qwen: Qwen3 235B A22B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.45499999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000455" } ], "output": [ { "amount": 1.8199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000182" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen3-235b-a22b", "canonicalSlug": "qwen/qwen3-235b-a22b-04-28", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 131072, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen3", "instruct_type": "qwen3" }, "description": "Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen3-235b-a22b-04-28", "model_id": "qwen/qwen3-235b-a22b", "model_name": "Qwen: Qwen3 235B A22B", "context_length": 131072, "pricing": { "prompt": "0.000000455", "completion": "0.00000182", "discount": 0 }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": 98304, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99662657918262, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o4-mini-high@openai", "name": "OpenAI: o4 Mini High", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o4-mini-high", "canonicalSlug": "openai/o4-mini-high-2025-04-16", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o4-mini-high-2025-04-16", "model_id": "openai/o4-mini-high", "model_name": "OpenAI: o4 Mini High", "context_length": 200000, "pricing": { "prompt": "0.0000011", "completion": "0.0000044", "web_search": "0.01", "input_cache_read": "0.000000275", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o4-mini-high:batch@openai", "name": "OpenAI: o4 Mini High", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.1375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001375" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o4-mini-high:batch", "canonicalSlug": "openai/o4-mini-high-2025-04-16", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o4-mini-high-2025-04-16:batch", "model_id": "openai/o4-mini-high:batch", "model_name": "OpenAI: o4 Mini High", "context_length": 200000, "pricing": { "prompt": "0.00000055", "completion": "0.0000022", "web_search": "0.01", "input_cache_read": "0.0000001375", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o3@openai", "name": "OpenAI: o3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o3", "canonicalSlug": "openai/o3-2025-04-16", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o3-2025-04-16", "model_id": "openai/o3", "model_name": "OpenAI: o3", "context_length": 200000, "pricing": { "prompt": "0.000002", "completion": "0.000008", "web_search": "0.01", "input_cache_read": "0.0000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o3:batch@openai", "name": "OpenAI: o3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o3:batch", "canonicalSlug": "openai/o3-2025-04-16", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o3-2025-04-16:batch", "model_id": "openai/o3:batch", "model_name": "OpenAI: o3", "context_length": 200000, "pricing": { "prompt": "0.000001", "completion": "0.000004", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o4-mini@openai", "name": "OpenAI: o4 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o4-mini", "canonicalSlug": "openai/o4-mini-2025-04-16", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o4-mini-2025-04-16", "model_id": "openai/o4-mini", "model_name": "OpenAI: o4 Mini", "context_length": 200000, "pricing": { "prompt": "0.0000011", "completion": "0.0000044", "web_search": "0.01", "input_cache_read": "0.000000275", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o4-mini:batch@openai", "name": "OpenAI: o4 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.1375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001375" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o4-mini:batch", "canonicalSlug": "openai/o4-mini-2025-04-16", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o4-mini-2025-04-16:batch", "model_id": "openai/o4-mini:batch", "model_name": "OpenAI: o4 Mini", "context_length": 200000, "pricing": { "prompt": "0.00000055", "completion": "0.0000022", "web_search": "0.01", "input_cache_read": "0.0000001375", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "structured_outputs", "response_format", "seed", "max_tokens", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1@azure", "name": "OpenAI: GPT-4.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1", "canonicalSlug": "openai/gpt-4.1-2025-04-14", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1047576, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4.1-2025-04-14", "model_id": "openai/gpt-4.1", "model_name": "OpenAI: GPT-4.1", "context_length": 1047576, "pricing": { "prompt": "0.000002", "completion": "0.000008", "web_search": "0.01", "input_cache_read": "0.0000005", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99720417882072, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1@openai", "name": "OpenAI: GPT-4.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1", "canonicalSlug": "openai/gpt-4.1-2025-04-14", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1047576, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...", "uptimeLast30m": 99.93814317044368, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4.1-2025-04-14", "model_id": "openai/gpt-4.1", "model_name": "OpenAI: GPT-4.1", "context_length": 1047576, "pricing": { "prompt": "0.000002", "completion": "0.000008", "web_search": "0.01", "input_cache_read": "0.0000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "status": 0, "uptime_last_30m": 99.93814317044368, "uptime_last_5m": 100, "uptime_last_1d": 99.82926325604764, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1@azure", "name": "OpenAI: GPT-4.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 8.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000088" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1", "canonicalSlug": "openai/gpt-4.1-2025-04-14", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1047576, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4.1-2025-04-14", "model_id": "openai/gpt-4.1", "model_name": "OpenAI: GPT-4.1", "context_length": 1047576, "pricing": { "prompt": "0.0000022", "completion": "0.0000088", "web_search": "0.01", "input_cache_read": "0.00000055", "discount": 0 }, "provider_name": "Azure", "tag": "azure/swedencentral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1:batch@openai", "name": "OpenAI: GPT-4.1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1:batch", "canonicalSlug": "openai/gpt-4.1-2025-04-14", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1047576, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4.1-2025-04-14:batch", "model_id": "openai/gpt-4.1:batch", "model_name": "OpenAI: GPT-4.1", "context_length": 1047576, "pricing": { "prompt": "0.000001", "completion": "0.000004", "web_search": "0.01", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1-mini@openai", "name": "OpenAI: GPT-4.1 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1-mini", "canonicalSlug": "openai/gpt-4.1-mini-2025-04-14", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1047576, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...", "uptimeLast30m": 99.94338994338995, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4.1-mini-2025-04-14", "model_id": "openai/gpt-4.1-mini", "model_name": "OpenAI: GPT-4.1 Mini", "context_length": 1047576, "pricing": { "prompt": "0.0000004", "completion": "0.0000016", "web_search": "0.01", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "status": 0, "uptime_last_30m": 99.94338994338995, "uptime_last_5m": 99.96791100652476, "uptime_last_1d": 98.51244200231582, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1-mini@azure", "name": "OpenAI: GPT-4.1 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1-mini", "canonicalSlug": "openai/gpt-4.1-mini-2025-04-14", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1047576, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4.1-mini-2025-04-14", "model_id": "openai/gpt-4.1-mini", "model_name": "OpenAI: GPT-4.1 Mini", "context_length": 1047576, "pricing": { "prompt": "0.0000004", "completion": "0.0000016", "web_search": "0.01", "input_cache_read": "0.0000001", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99164705362101, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1-mini@azure", "name": "OpenAI: GPT-4.1 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000044" } ], "output": [ { "amount": 1.76, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000176" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1-mini", "canonicalSlug": "openai/gpt-4.1-mini-2025-04-14", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1047576, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4.1-mini-2025-04-14", "model_id": "openai/gpt-4.1-mini", "model_name": "OpenAI: GPT-4.1 Mini", "context_length": 1047576, "pricing": { "prompt": "0.00000044", "completion": "0.00000176", "web_search": "0.01", "input_cache_read": "0.00000011", "discount": 0 }, "provider_name": "Azure", "tag": "azure/swedencentral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1-mini:batch@openai", "name": "OpenAI: GPT-4.1 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1-mini:batch", "canonicalSlug": "openai/gpt-4.1-mini-2025-04-14", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1047576, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4.1-mini-2025-04-14:batch", "model_id": "openai/gpt-4.1-mini:batch", "model_name": "OpenAI: GPT-4.1 Mini", "context_length": 1047576, "pricing": { "prompt": "0.0000002", "completion": "0.0000008", "web_search": "0.01", "input_cache_read": "0.00000005", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1-nano@openai", "name": "OpenAI: GPT-4.1 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1-nano", "canonicalSlug": "openai/gpt-4.1-nano-2025-04-14", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1047576, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...", "uptimeLast30m": 99.50134061693437, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4.1-nano-2025-04-14", "model_id": "openai/gpt-4.1-nano", "model_name": "OpenAI: GPT-4.1 Nano", "context_length": 1047576, "pricing": { "prompt": "0.0000001", "completion": "0.0000004", "web_search": "0.01", "input_cache_read": "0.000000025", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "status": 0, "uptime_last_30m": 99.50134061693437, "uptime_last_5m": 99.97948717948718, "uptime_last_1d": 99.45521310096723, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1-nano@azure", "name": "OpenAI: GPT-4.1 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1-nano", "canonicalSlug": "openai/gpt-4.1-nano-2025-04-14", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1047576, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...", "uptimeLast30m": 99.8936170212766, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4.1-nano-2025-04-14", "model_id": "openai/gpt-4.1-nano", "model_name": "OpenAI: GPT-4.1 Nano", "context_length": 1047576, "pricing": { "prompt": "0.0000001", "completion": "0.0000004", "web_search": "0.01", "input_cache_read": "0.00000003", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "temperature", "top_p" ], "status": 0, "uptime_last_30m": 99.8936170212766, "uptime_last_5m": 100, "uptime_last_1d": 99.88427964076375, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1-nano@azure", "name": "OpenAI: GPT-4.1 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "output": [ { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000044" } ], "cacheRead": [ { "amount": 0.032999999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000033" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1-nano", "canonicalSlug": "openai/gpt-4.1-nano-2025-04-14", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 1047576, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4.1-nano-2025-04-14", "model_id": "openai/gpt-4.1-nano", "model_name": "OpenAI: GPT-4.1 Nano", "context_length": 1047576, "pricing": { "prompt": "0.00000011", "completion": "0.00000044", "web_search": "0.01", "input_cache_read": "0.000000033", "discount": 0 }, "provider_name": "Azure", "tag": "azure/swedencentral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": true, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4.1-nano:batch@openai", "name": "OpenAI: GPT-4.1 Nano", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [ { "amount": 0.012499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000125" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4.1-nano:batch", "canonicalSlug": "openai/gpt-4.1-nano-2025-04-14", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 1047576, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "image", "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4.1-nano-2025-04-14:batch", "model_id": "openai/gpt-4.1-nano:batch", "model_name": "OpenAI: GPT-4.1 Nano", "context_length": 1047576, "pricing": { "prompt": "0.00000005", "completion": "0.0000002", "web_search": "0.01", "input_cache_read": "0.0000000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-4-maverick@digitalocean", "name": "Meta: Llama 4 Maverick", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.696, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000696" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-4-maverick", "canonicalSlug": "meta-llama/llama-4-maverick-17b-128e-instruct", "servingProvider": "DigitalOcean", "servingProviderSlug": "digitalocean", "contextLength": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "temperature", "top_p", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Llama4", "instruct_type": null }, "description": "Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DigitalOcean | meta-llama/llama-4-maverick-17b-128e-instruct", "model_id": "meta-llama/llama-4-maverick", "model_name": "Meta: Llama 4 Maverick", "context_length": 128000, "pricing": { "prompt": "0.0000002", "completion": "0.000000696", "discount": 0 }, "provider_name": "DigitalOcean", "tag": "digitalocean", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "temperature", "top_p", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 98.7551867219917, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-4-maverick@deepinfra", "name": "Meta: Llama 4 Maverick", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-4-maverick", "canonicalSlug": "meta-llama/llama-4-maverick-17b-128e-instruct", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 1048576, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Llama4", "instruct_type": null }, "description": "Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...", "uptimeLast30m": 99.99478542003442, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | meta-llama/llama-4-maverick-17b-128e-instruct", "model_id": "meta-llama/llama-4-maverick", "model_name": "Meta: Llama 4 Maverick", "context_length": 1048576, "pricing": { "prompt": "0.0000002", "completion": "0.0000008", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/base", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "logit_bias" ], "status": 0, "uptime_last_30m": 99.99478542003442, "uptime_last_5m": 100, "uptime_last_1d": 99.72809330284093, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-4-maverick@novita", "name": "Meta: Llama 4 Maverick", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000085" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-4-maverick", "canonicalSlug": "meta-llama/llama-4-maverick-17b-128e-instruct", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 1048576, "maxCompletionTokens": 8192, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "structured_outputs", "response_format" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Llama4", "instruct_type": null }, "description": "Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | meta-llama/llama-4-maverick-17b-128e-instruct", "model_id": "meta-llama/llama-4-maverick", "model_name": "Meta: Llama 4 Maverick", "context_length": 1048576, "pricing": { "prompt": "0.00000027", "completion": "0.00000085", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.69276963913975, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-4-maverick@parasail", "name": "Meta: Llama 4 Maverick", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000017" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-4-maverick", "canonicalSlug": "meta-llama/llama-4-maverick-17b-128e-instruct", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 524288, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Llama4", "instruct_type": null }, "description": "Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | meta-llama/llama-4-maverick-17b-128e-instruct", "model_id": "meta-llama/llama-4-maverick", "model_name": "Meta: Llama 4 Maverick", "context_length": 524288, "pricing": { "prompt": "0.00000035", "completion": "0.000001", "input_cache_read": "0.00000017", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96886857841007, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-4-maverick@google", "name": "Meta: Llama 4 Maverick", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000035" } ], "output": [ { "amount": 1.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000115" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-4-maverick", "canonicalSlug": "meta-llama/llama-4-maverick-17b-128e-instruct", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 524288, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Llama4", "instruct_type": null }, "description": "Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...", "uptimeLast30m": 99.74025974025975, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | meta-llama/llama-4-maverick-17b-128e-instruct", "model_id": "meta-llama/llama-4-maverick", "model_name": "Meta: Llama 4 Maverick", "context_length": 524288, "pricing": { "prompt": "0.00000035", "completion": "0.00000115", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us-east5", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.74025974025975, "uptime_last_5m": 100, "uptime_last_1d": 99.89589267123691, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-4-scout@deepinfra", "name": "Meta: Llama 4 Scout", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-4-scout", "canonicalSlug": "meta-llama/llama-4-scout-17b-16e-instruct", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 327680, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Llama4", "instruct_type": null }, "description": "Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | meta-llama/llama-4-scout-17b-16e-instruct", "model_id": "meta-llama/llama-4-scout", "model_name": "Meta: Llama 4 Scout", "context_length": 327680, "pricing": { "prompt": "0.0000001", "completion": "0.0000003", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.4807811921076, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-4-scout@groq", "name": "Meta: Llama 4 Scout", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "output": [ { "amount": 0.33999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000034" } ], "cacheRead": [ { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000055" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-4-scout", "canonicalSlug": "meta-llama/llama-4-scout-17b-16e-instruct", "servingProvider": "Groq", "servingProviderSlug": "groq", "contextLength": 131072, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Llama4", "instruct_type": null }, "description": "Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...", "uptimeLast30m": 99.96568291008923, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Groq | meta-llama/llama-4-scout-17b-16e-instruct", "model_id": "meta-llama/llama-4-scout", "model_name": "Meta: Llama 4 Scout", "context_length": 131072, "pricing": { "prompt": "0.00000011", "completion": "0.00000034", "input_cache_read": "0.000000055", "discount": 0 }, "provider_name": "Groq", "tag": "groq", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.96568291008923, "uptime_last_5m": 100, "uptime_last_1d": 99.83941907209753, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-4-scout@novita", "name": "Meta: Llama 4 Scout", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000018" } ], "output": [ { "amount": 0.59, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000059" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-4-scout", "canonicalSlug": "meta-llama/llama-4-scout-17b-16e-instruct", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Llama4", "instruct_type": null }, "description": "Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | meta-llama/llama-4-scout-17b-16e-instruct", "model_id": "meta-llama/llama-4-scout", "model_name": "Meta: Llama 4 Scout", "context_length": 131072, "pricing": { "prompt": "0.00000018", "completion": "0.00000059", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.75420174861739, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-4-scout@google", "name": "Meta: Llama 4 Scout", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-4-scout", "canonicalSlug": "meta-llama/llama-4-scout-17b-16e-instruct", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 1310720, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Llama4", "instruct_type": null }, "description": "Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | meta-llama/llama-4-scout-17b-16e-instruct", "model_id": "meta-llama/llama-4-scout", "model_name": "Meta: Llama 4 Scout", "context_length": 1310720, "pricing": { "prompt": "0.00000025", "completion": "0.0000007", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us-east5", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.9303102006418, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3-0324@deepinfra", "name": "DeepSeek: DeepSeek V3 0324", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000024" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000135" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3-0324", "canonicalSlug": "deepseek/deepseek-chat-v3-0324", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 163840, "maxCompletionTokens": 16384, "quantization": "fp4", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...", "uptimeLast30m": 97.99851742031134, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | deepseek/deepseek-chat-v3-0324", "model_id": "deepseek/deepseek-chat-v3-0324", "model_name": "DeepSeek: DeepSeek V3 0324", "context_length": 163840, "pricing": { "prompt": "0.00000024", "completion": "0.0000009", "input_cache_read": "0.000000135", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": 97.99851742031134, "uptime_last_5m": 100, "uptime_last_1d": 93.70220481959373, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3-0324@siliconflow", "name": "DeepSeek: DeepSeek V3 0324", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3-0324", "canonicalSlug": "deepseek/deepseek-chat-v3-0324", "servingProvider": "SiliconFlow", "servingProviderSlug": "siliconflow", "contextLength": 163840, "maxCompletionTokens": 163840, "quantization": "fp8", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...", "uptimeLast30m": 98.39498299319727, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SiliconFlow | deepseek/deepseek-chat-v3-0324", "model_id": "deepseek/deepseek-chat-v3-0324", "model_name": "DeepSeek: DeepSeek V3 0324", "context_length": 163840, "pricing": { "prompt": "0.00000025", "completion": "0.000001", "discount": 0 }, "provider_name": "SiliconFlow", "tag": "siliconflow/fp8", "quantization": "fp8", "max_completion_tokens": 163840, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "temperature", "top_p", "top_k", "frequency_penalty", "tools", "tool_choice", "max_tokens" ], "status": 0, "uptime_last_30m": 98.39498299319727, "uptime_last_5m": 97.0013037809648, "uptime_last_1d": 96.97909498841841, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3-0324@novita", "name": "DeepSeek: DeepSeek V3 0324", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000027" } ], "output": [ { "amount": 1.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000112" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000135" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3-0324", "canonicalSlug": "deepseek/deepseek-chat-v3-0324", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 163840, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...", "uptimeLast30m": 99.0401023890785, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-chat-v3-0324", "model_id": "deepseek/deepseek-chat-v3-0324", "model_name": "DeepSeek: DeepSeek V3 0324", "context_length": 163840, "pricing": { "prompt": "0.00000027", "completion": "0.00000112", "input_cache_read": "0.000000135", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.0401023890785, "uptime_last_5m": 100, "uptime_last_1d": 97.98599902065689, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat-v3-0324@crusoe", "name": "DeepSeek: DeepSeek V3 0324", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat-v3-0324", "canonicalSlug": "deepseek/deepseek-chat-v3-0324", "servingProvider": "Crusoe", "servingProviderSlug": "crusoe", "contextLength": 163840, "maxCompletionTokens": 163840, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Crusoe | deepseek/deepseek-chat-v3-0324", "model_id": "deepseek/deepseek-chat-v3-0324", "model_name": "DeepSeek: DeepSeek V3 0324", "context_length": 163840, "pricing": { "prompt": "0.0000005", "completion": "0.0000015", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "Crusoe", "tag": "crusoe/bf16", "quantization": "bf16", "max_completion_tokens": 163840, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9337774142028, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o1-pro@openai", "name": "OpenAI: o1-pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 150, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00015" } ], "output": [ { "amount": 600, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0006" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o1-pro", "canonicalSlug": "openai/o1-pro", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o1-pro", "model_id": "openai/o1-pro", "model_name": "OpenAI: o1-pro", "context_length": 200000, "pricing": { "prompt": "0.00015", "completion": "0.0006", "web_search": "0.01", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o1-pro:batch@openai", "name": "OpenAI: o1-pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "output": [ { "amount": 300, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0003" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o1-pro:batch", "canonicalSlug": "openai/o1-pro", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o1-pro:batch", "model_id": "openai/o1-pro:batch", "model_name": "OpenAI: o1-pro", "context_length": 200000, "pricing": { "prompt": "0.000075", "completion": "0.0003", "web_search": "0.01", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-small-3.1-24b-instruct@cloudflare", "name": "Mistral: Mistral Small 3.1 24B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.351, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000351" } ], "output": [ { "amount": 0.5549999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000555" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-small-3.1-24b-instruct", "canonicalSlug": "mistralai/mistral-small-3.1-24b-instruct-2503", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 128000, "maxCompletionTokens": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | mistralai/mistral-small-3.1-24b-instruct-2503", "model_id": "mistralai/mistral-small-3.1-24b-instruct", "model_name": "Mistral: Mistral Small 3.1 24B", "context_length": 128000, "pricing": { "prompt": "0.000000351", "completion": "0.000000555", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99810098938453, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-3-4b-it@deepinfra", "name": "Google: Gemma 3 4B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-3-4b-it", "canonicalSlug": "google/gemma-3-4b-it", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...", "uptimeLast30m": 99.24528301886792, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | google/gemma-3-4b-it", "model_id": "google/gemma-3-4b-it", "model_name": "Google: Gemma 3 4B", "context_length": 131072, "pricing": { "prompt": "0.00000005", "completion": "0.0000001", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.24528301886792, "uptime_last_5m": null, "uptime_last_1d": 99.87724236155854, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-3-12b-it@deepinfra", "name": "Google: Gemma 3 12B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-3-12b-it", "canonicalSlug": "google/gemma-3-12b-it", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | google/gemma-3-12b-it", "model_id": "google/gemma-3-12b-it", "model_name": "Google: Gemma 3 12B", "context_length": 131072, "pricing": { "prompt": "0.00000005", "completion": "0.00000015", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.75237569496471, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "cohere/command-a@cohere", "name": "Cohere: Command A", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "cohere/command-a", "canonicalSlug": "cohere/command-a-03-2025", "servingProvider": "Cohere", "servingProviderSlug": "cohere", "contextLength": 256000, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cohere | cohere/command-a-03-2025", "model_id": "cohere/command-a", "model_name": "Cohere: Command A", "context_length": 256000, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "discount": 0 }, "provider_name": "Cohere", "tag": "cohere", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.86103068905616, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "rekaai/reka-flash-3@reka", "name": "Reka Flash 3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "rekaai/reka-flash-3", "canonicalSlug": "rekaai/reka-flash-3", "servingProvider": "Reka", "servingProviderSlug": "reka", "contextLength": 65536, "maxCompletionTokens": 65536, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "max_tokens", "stop", "seed", "frequency_penalty", "presence_penalty", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Reka | rekaai/reka-flash-3", "model_id": "rekaai/reka-flash-3", "model_name": "Reka Flash 3", "context_length": 65536, "pricing": { "prompt": "0.0000001", "completion": "0.0000002", "discount": 0 }, "provider_name": "Reka", "tag": "reka/fp8", "quantization": "fp8", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "temperature", "top_p", "top_k", "max_tokens", "stop", "seed", "frequency_penalty", "presence_penalty", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-3-27b-it@deepinfra", "name": "Google: Gemma 3 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "output": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-3-27b-it", "canonicalSlug": "google/gemma-3-27b-it", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...", "uptimeLast30m": 99.95551931203202, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | google/gemma-3-27b-it", "model_id": "google/gemma-3-27b-it", "model_name": "Google: Gemma 3 27B", "context_length": 131072, "pricing": { "prompt": "0.00000008", "completion": "0.00000016", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.95551931203202, "uptime_last_5m": 99.66216216216216, "uptime_last_1d": 99.56077702093037, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-3-27b-it@parasail", "name": "Google: Gemma 3 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-3-27b-it", "canonicalSlug": "google/gemma-3-27b-it", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...", "uptimeLast30m": 99.23132012413464, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | google/gemma-3-27b-it", "model_id": "google/gemma-3-27b-it", "model_name": "Google: Gemma 3 27B", "context_length": 131072, "pricing": { "prompt": "0.00000008", "completion": "0.00000045", "input_cache_read": "0.00000004", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.23132012413464, "uptime_last_5m": 98.73932492883286, "uptime_last_1d": 99.6819388375664, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-3-27b-it@nebius", "name": "Google: Gemma 3 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-3-27b-it", "canonicalSlug": "google/gemma-3-27b-it", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 110000, "quantization": "fp8", "status": -2, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "structured_outputs", "response_format", "repetition_penalty" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...", "uptimeLast30m": 88.04429901837403, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | google/gemma-3-27b-it", "model_id": "google/gemma-3-27b-it", "model_name": "Google: Gemma 3 27B", "context_length": 110000, "pricing": { "prompt": "0.0000001", "completion": "0.0000003", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "structured_outputs", "response_format", "repetition_penalty" ], "status": -2, "uptime_last_30m": 88.04429901837403, "uptime_last_5m": 94.45438282647585, "uptime_last_1d": 89.85689740618112, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-3-27b-it@novita", "name": "Google: Gemma 3 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.119, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000119" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-3-27b-it", "canonicalSlug": "google/gemma-3-27b-it", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 98304, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...", "uptimeLast30m": 99.09064241713112, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | google/gemma-3-27b-it", "model_id": "google/gemma-3-27b-it", "model_name": "Google: Gemma 3 27B", "context_length": 98304, "pricing": { "prompt": "0.000000119", "completion": "0.0000002", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty" ], "status": 0, "uptime_last_30m": 99.09064241713112, "uptime_last_5m": 99.55456570155901, "uptime_last_1d": 96.2261599071031, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-3-27b-it@phala", "name": "Google: Gemma 3 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.45999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000046" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-3-27b-it", "canonicalSlug": "google/gemma-3-27b-it", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 262144, "maxCompletionTokens": 262144, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | google/gemma-3-27b-it", "model_id": "google/gemma-3-27b-it", "model_name": "Google: Gemma 3 27B", "context_length": 262144, "pricing": { "prompt": "0.00000015", "completion": "0.00000046", "input_cache_read": "0.000000075", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 262144, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 93.58624454148472, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thedrummer/skyfall-36b-v2@parasail", "name": "TheDrummer: Skyfall 36B V2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thedrummer/skyfall-36b-v2", "canonicalSlug": "thedrummer/skyfall-36b-v2", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 32768, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | thedrummer/skyfall-36b-v2", "model_id": "thedrummer/skyfall-36b-v2", "model_name": "TheDrummer: Skyfall 36B V2", "context_length": 32768, "pricing": { "prompt": "0.00000055", "completion": "0.0000008", "input_cache_read": "0.00000025", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "perplexity/sonar-reasoning-pro@perplexity", "name": "Perplexity: Sonar Reasoning Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "perplexity/sonar-reasoning-pro", "canonicalSlug": "perplexity/sonar-reasoning-pro", "servingProvider": "Perplexity", "servingProviderSlug": "perplexity", "contextLength": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": "deepseek-r1" }, "description": "Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Perplexity | perplexity/sonar-reasoning-pro", "model_id": "perplexity/sonar-reasoning-pro", "model_name": "Perplexity: Sonar Reasoning Pro", "context_length": 128000, "pricing": { "prompt": "0.000002", "completion": "0.000008", "web_search": "0.005", "discount": 0 }, "provider_name": "Perplexity", "tag": "perplexity", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.85326485693324, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "perplexity/sonar-pro@perplexity", "name": "Perplexity: Sonar Pro", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "perplexity/sonar-pro", "canonicalSlug": "perplexity/sonar-pro", "servingProvider": "Perplexity", "servingProviderSlug": "perplexity", "contextLength": 200000, "maxCompletionTokens": 8000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Perplexity | perplexity/sonar-pro", "model_id": "perplexity/sonar-pro", "model_name": "Perplexity: Sonar Pro", "context_length": 200000, "pricing": { "prompt": "0.000003", "completion": "0.000015", "web_search": "0.005", "discount": 0 }, "provider_name": "Perplexity", "tag": "perplexity", "quantization": "unknown", "max_completion_tokens": 8000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.98616081137186, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "perplexity/sonar-deep-research@perplexity", "name": "Perplexity: Sonar Deep Research", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" }, { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ] } ], "metadata": { "source": "openrouter", "modelId": "perplexity/sonar-deep-research", "canonicalSlug": "perplexity/sonar-deep-research", "servingProvider": "Perplexity", "servingProviderSlug": "perplexity", "contextLength": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": "deepseek-r1" }, "description": "Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Perplexity | perplexity/sonar-deep-research", "model_id": "perplexity/sonar-deep-research", "model_name": "Perplexity: Sonar Deep Research", "context_length": 128000, "pricing": { "prompt": "0.000002", "completion": "0.000008", "web_search": "0.005", "internal_reasoning": "0.000003", "discount": 0 }, "provider_name": "Perplexity", "tag": "perplexity", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-saba@mistral", "name": "Mistral: Saba", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-saba", "canonicalSlug": "mistralai/mistral-saba-2502", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/mistral-saba-2502", "model_id": "mistralai/mistral-saba", "model_name": "Mistral: Saba", "context_length": 32768, "pricing": { "prompt": "0.0000002", "completion": "0.0000006", "input_cache_read": "0.00000002", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o3-mini-high@openai", "name": "OpenAI: o3 Mini High", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o3-mini-high", "canonicalSlug": "openai/o3-mini-high-2025-01-31", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o3-mini-high-2025-01-31", "model_id": "openai/o3-mini-high", "model_name": "OpenAI: o3 Mini High", "context_length": 200000, "pricing": { "prompt": "0.0000011", "completion": "0.0000044", "web_search": "0.01", "input_cache_read": "0.00000055", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o3-mini-high:batch@openai", "name": "OpenAI: o3 Mini High", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o3-mini-high:batch", "canonicalSlug": "openai/o3-mini-high-2025-01-31", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o3-mini-high-2025-01-31:batch", "model_id": "openai/o3-mini-high:batch", "model_name": "OpenAI: o3 Mini High", "context_length": 200000, "pricing": { "prompt": "0.00000055", "completion": "0.0000022", "web_search": "0.01", "input_cache_read": "0.000000275", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "aion-labs/aion-rp-llama-3.1-8b@aionlabs", "name": "AionLabs: Aion-RP 1.0 (8B)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "aion-labs/aion-rp-llama-3.1-8b", "canonicalSlug": "aion-labs/aion-rp-llama-3.1-8b", "servingProvider": "AionLabs", "servingProviderSlug": "aionlabs", "contextLength": 32768, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AionLabs | aion-labs/aion-rp-llama-3.1-8b", "model_id": "aion-labs/aion-rp-llama-3.1-8b", "model_name": "AionLabs: Aion-RP 1.0 (8B)", "context_length": 32768, "pricing": { "prompt": "0.0000008", "completion": "0.0000016", "discount": 0 }, "provider_name": "AionLabs", "tag": "aion-labs", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen2.5-vl-72b-instruct@nebius", "name": "Qwen: Qwen2.5 VL 72B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen2.5-vl-72b-instruct", "canonicalSlug": "qwen/qwen2.5-vl-72b-instruct", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 32000, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | qwen/qwen2.5-vl-72b-instruct", "model_id": "qwen/qwen2.5-vl-72b-instruct", "model_name": "Qwen: Qwen2.5 VL 72B Instruct", "context_length": 32000, "pricing": { "prompt": "0.00000025", "completion": "0.00000075", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 93.83453499541787, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen2.5-vl-72b-instruct@parasail", "name": "Qwen: Qwen2.5 VL 72B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen2.5-vl-72b-instruct", "canonicalSlug": "qwen/qwen2.5-vl-72b-instruct", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 128000, "maxCompletionTokens": 128000, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.", "uptimeLast30m": 99.94717379820392, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | qwen/qwen2.5-vl-72b-instruct", "model_id": "qwen/qwen2.5-vl-72b-instruct", "model_name": "Qwen: Qwen2.5 VL 72B Instruct", "context_length": 128000, "pricing": { "prompt": "0.0000008", "completion": "0.000001", "input_cache_read": "0.0000004", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.94717379820392, "uptime_last_5m": 100, "uptime_last_1d": 99.73127894712707, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-plus@alibaba", "name": "Qwen: Qwen-Plus", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000026" } ], "output": [ { "amount": 0.78, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000078" } ], "cacheRead": [ { "amount": 0.052000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000052" } ], "cacheWrite": [ { "amount": 0.325, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000325" } ], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-plus", "canonicalSlug": "qwen/qwen-plus-2025-01-25", "servingProvider": "Alibaba", "servingProviderSlug": "alibaba", "contextLength": 1000000, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": null }, "description": "Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Alibaba | qwen/qwen-plus-2025-01-25", "model_id": "qwen/qwen-plus", "model_name": "Qwen: Qwen-Plus", "context_length": 1000000, "pricing": { "prompt": "0.00000026", "completion": "0.00000078", "input_cache_read": "0.000000052", "input_cache_write": "0.000000325", "discount": 0, "overrides": [ { "min_prompt_tokens": 256000, "prompt": "0.00000078", "completion": "0.00000234", "input_cache_read": "0.000000156", "input_cache_write": "0.000000975" } ] }, "provider_name": "Alibaba", "tag": "alibaba", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": 995904, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "tools", "tool_choice", "structured_outputs", "logprobs", "top_logprobs", "top_k", "frequency_penalty", "stop" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o3-mini@openai", "name": "OpenAI: o3 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o3-mini", "canonicalSlug": "openai/o3-mini-2025-01-31", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o3-mini-2025-01-31", "model_id": "openai/o3-mini", "model_name": "OpenAI: o3 Mini", "context_length": 200000, "pricing": { "prompt": "0.0000011", "completion": "0.0000044", "web_search": "0.01", "input_cache_read": "0.00000055", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o3-mini:batch@openai", "name": "OpenAI: o3 Mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000055" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000275" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o3-mini:batch", "canonicalSlug": "openai/o3-mini-2025-01-31", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o3-mini-2025-01-31:batch", "model_id": "openai/o3-mini:batch", "model_name": "OpenAI: o3 Mini", "context_length": 200000, "pricing": { "prompt": "0.00000055", "completion": "0.0000022", "web_search": "0.01", "input_cache_read": "0.000000275", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-small-24b-instruct-2501@deepinfra", "name": "Mistral: Mistral Small 3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-small-24b-instruct-2501", "canonicalSlug": "mistralai/mistral-small-24b-instruct-2501", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 32768, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | mistralai/mistral-small-24b-instruct-2501", "model_id": "mistralai/mistral-small-24b-instruct-2501", "model_name": "Mistral: Mistral Small 3", "context_length": 32768, "pricing": { "prompt": "0.00000005", "completion": "0.00000008", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99933174436994, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "perplexity/sonar@perplexity", "name": "Perplexity: Sonar", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "openrouter", "modelId": "perplexity/sonar", "canonicalSlug": "perplexity/sonar", "servingProvider": "Perplexity", "servingProviderSlug": "perplexity", "contextLength": 127072, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Perplexity | perplexity/sonar", "model_id": "perplexity/sonar", "model_name": "Perplexity: Sonar", "context_length": 127072, "pricing": { "prompt": "0.000001", "completion": "0.000001", "web_search": "0.005", "discount": 0 }, "provider_name": "Perplexity", "tag": "perplexity", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "frequency_penalty", "presence_penalty", "web_search_options" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99569715946328, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-r1-distill-llama-70b@novita", "name": "DeepSeek: R1 Distill Llama 70B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-r1-distill-llama-70b", "canonicalSlug": "deepseek/deepseek-r1-distill-llama-70b", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 8192, "maxCompletionTokens": 8192, "quantization": "bf16", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "deepseek-r1" }, "description": "DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-r1-distill-llama-70b", "model_id": "deepseek/deepseek-r1-distill-llama-70b", "model_name": "DeepSeek: R1 Distill Llama 70B", "context_length": 8192, "pricing": { "prompt": "0.0000008", "completion": "0.0000008", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-r1@novita", "name": "DeepSeek: R1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-r1", "canonicalSlug": "deepseek/deepseek-r1", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 64000, "maxCompletionTokens": 16000, "quantization": "fp8", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": "deepseek-r1" }, "description": "DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-r1", "model_id": "deepseek/deepseek-r1", "model_name": "DeepSeek: R1", "context_length": 64000, "pricing": { "prompt": "0.0000007", "completion": "0.0000025", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 16000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "minimax/minimax-01@minimax", "name": "MiniMax: MiniMax-01", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "minimax/minimax-01", "canonicalSlug": "minimax/minimax-01", "servingProvider": "Minimax", "servingProviderSlug": "minimax", "contextLength": 1000192, "maxCompletionTokens": 1000192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Minimax | minimax/minimax-01", "model_id": "minimax/minimax-01", "model_name": "MiniMax: MiniMax-01", "context_length": 1000192, "pricing": { "prompt": "0.0000002", "completion": "0.0000011", "discount": 0 }, "provider_name": "Minimax", "tag": "minimax", "quantization": "unknown", "max_completion_tokens": 1000192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "microsoft/phi-4@deepinfra", "name": "Microsoft: Phi 4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000007" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "microsoft/phi-4", "canonicalSlug": "microsoft/phi-4", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 16384, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Other", "instruct_type": null }, "description": "[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | microsoft/phi-4", "model_id": "microsoft/phi-4", "model_name": "Microsoft: Phi 4", "context_length": 16384, "pricing": { "prompt": "0.00000007", "completion": "0.00000014", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": 4096, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99838181156196, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat@streamlake", "name": "DeepSeek: DeepSeek V3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.2574, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002574" } ], "output": [ { "amount": 1.0287, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000010287" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat", "canonicalSlug": "deepseek/deepseek-chat-v3", "servingProvider": "StreamLake", "servingProviderSlug": "streamlake", "contextLength": 128000, "maxCompletionTokens": 16000, "quantization": "unknown", "status": 0, "supportedParameters": [ "response_format", "structured_outputs", "tools", "tool_choice", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...", "uptimeLast30m": 99.80266792959192, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "StreamLake | deepseek/deepseek-chat-v3", "model_id": "deepseek/deepseek-chat", "model_name": "DeepSeek: DeepSeek V3", "context_length": 128000, "pricing": { "prompt": "0.0000002574", "completion": "0.0000010287", "discount": 0.1 }, "provider_name": "StreamLake", "tag": "streamlake", "quantization": "unknown", "max_completion_tokens": 16000, "max_prompt_tokens": null, "supported_parameters": [ "response_format", "structured_outputs", "tools", "tool_choice", "max_tokens", "temperature", "top_p", "stop", "presence_penalty" ], "status": 0, "uptime_last_30m": 99.80266792959192, "uptime_last_5m": 100, "uptime_last_1d": 99.35252742373326, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat@deepinfra", "name": "DeepSeek: DeepSeek V3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000032" } ], "output": [ { "amount": 0.8899999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000089" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat", "canonicalSlug": "deepseek/deepseek-chat-v3", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 163840, "maxCompletionTokens": 16384, "quantization": "fp4", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...", "uptimeLast30m": 99.07649684469754, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | deepseek/deepseek-chat-v3", "model_id": "deepseek/deepseek-chat", "model_name": "DeepSeek: DeepSeek V3", "context_length": 163840, "pricing": { "prompt": "0.00000032", "completion": "0.00000089", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp4", "quantization": "fp4", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.07649684469754, "uptime_last_5m": 99.76019184652279, "uptime_last_1d": 96.9426628306983, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "deepseek/deepseek-chat@novita", "name": "DeepSeek: DeepSeek V3", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 1.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000013" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "deepseek/deepseek-chat", "canonicalSlug": "deepseek/deepseek-chat-v3", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 64000, "maxCompletionTokens": 16000, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "DeepSeek", "instruct_type": null }, "description": "DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...", "uptimeLast30m": 99.98700285937095, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | deepseek/deepseek-chat-v3", "model_id": "deepseek/deepseek-chat", "model_name": "DeepSeek: DeepSeek V3", "context_length": 64000, "pricing": { "prompt": "0.0000004", "completion": "0.0000013", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 16000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.98700285937095, "uptime_last_5m": 100, "uptime_last_1d": 99.9611607301138, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sao10k/l3.3-euryale-70b@nextbit", "name": "Sao10K: Llama 3.3 Euryale 70B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000065" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sao10k/l3.3-euryale-70b", "canonicalSlug": "sao10k/l3.3-euryale-70b-v2.3", "servingProvider": "NextBit", "servingProviderSlug": "nextbit", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "logprobs", "top_logprobs", "repetition_penalty", "seed" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "NextBit | sao10k/l3.3-euryale-70b-v2.3", "model_id": "sao10k/l3.3-euryale-70b", "model_name": "Sao10K: Llama 3.3 Euryale 70B", "context_length": 131072, "pricing": { "prompt": "0.00000065", "completion": "0.00000075", "discount": 0 }, "provider_name": "NextBit", "tag": "nextbit/bf16", "quantization": "bf16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "logprobs", "top_logprobs", "repetition_penalty", "seed" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 95.6921587608906, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o1@openai", "name": "OpenAI: o1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o1", "canonicalSlug": "openai/o1-2024-12-17", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o1-2024-12-17", "model_id": "openai/o1", "model_name": "OpenAI: o1", "context_length": 200000, "pricing": { "prompt": "0.000015", "completion": "0.00006", "web_search": "0.01", "input_cache_read": "0.0000075", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/o1:batch@openai", "name": "OpenAI: o1", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/o1:batch", "canonicalSlug": "openai/o1-2024-12-17", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 200000, "maxCompletionTokens": 100000, "quantization": "unknown", "status": 0, "supportedParameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/o1-2024-12-17:batch", "model_id": "openai/o1:batch", "model_name": "OpenAI: o1", "context_length": 200000, "pricing": { "prompt": "0.0000075", "completion": "0.00003", "web_search": "0.01", "input_cache_read": "0.00000375", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 100000, "max_prompt_tokens": null, "supported_parameters": [ "reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "cohere/command-r7b-12-2024@cohere", "name": "Cohere: Command R7B (12-2024)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "cohere/command-r7b-12-2024", "canonicalSlug": "cohere/command-r7b-12-2024", "servingProvider": "Cohere", "servingProviderSlug": "cohere", "contextLength": 128000, "maxCompletionTokens": 4000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Cohere", "instruct_type": null }, "description": "Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cohere | cohere/command-r7b-12-2024", "model_id": "cohere/command-r7b-12-2024", "model_name": "Cohere: Command R7B (12-2024)", "context_length": 128000, "pricing": { "prompt": "0.0000000375", "completion": "0.00000015", "discount": 0 }, "provider_name": "Cohere", "tag": "cohere", "quantization": "unknown", "max_completion_tokens": 4000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.91666666666667, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@deepinfra", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 97.11071640023683, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 131072, "pricing": { "prompt": "0.0000001", "completion": "0.00000032", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/turbo", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias" ], "status": 0, "uptime_last_30m": 97.11071640023683, "uptime_last_5m": 97.03632887189293, "uptime_last_1d": 97.598439313267, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@nebius", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "Nebius", "servingProviderSlug": "nebius", "contextLength": 131072, "quantization": "fp8", "status": -5, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "repetition_penalty" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 41.81360201511335, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Nebius | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000013", "completion": "0.0000004", "discount": 0 }, "provider_name": "Nebius", "tag": "nebius/fp8", "quantization": "fp8", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "logit_bias", "tools", "tool_choice", "response_format", "structured_outputs", "repetition_penalty" ], "status": -5, "uptime_last_30m": 41.81360201511335, "uptime_last_5m": 37.39424703891709, "uptime_last_1d": 49.617926681618805, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@akashml", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "AkashML", "servingProviderSlug": "akashml", "contextLength": 131072, "maxCompletionTokens": 128000, "quantization": "fp8", "status": 0, "supportedParameters": [ "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "max_tokens", "structured_outputs", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 97.31245618135078, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "AkashML | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000013", "completion": "0.0000004", "discount": 0 }, "provider_name": "AkashML", "tag": "akashml/fp8", "quantization": "fp8", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "max_tokens", "structured_outputs", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 97.31245618135078, "uptime_last_5m": 98.73551106427819, "uptime_last_1d": 96.01118239180533, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@novita", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000135" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 12288, "maxCompletionTokens": 12288, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 98.55177407675598, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 12288, "pricing": { "prompt": "0.000000135", "completion": "0.0000004", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 12288, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 98.55177407675598, "uptime_last_5m": 98.74213836477988, "uptime_last_1d": 95.99089807811937, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@parasail", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 97.33672603901611, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000022", "completion": "0.0000005", "input_cache_read": "0.00000011", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 97.33672603901611, "uptime_last_5m": 94.94640122511485, "uptime_last_1d": 95.56043992456227, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@crusoe", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000013" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "Crusoe", "servingProviderSlug": "crusoe", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 99.94058229352348, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Crusoe | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000025", "completion": "0.00000075", "input_cache_read": "0.00000013", "discount": 0 }, "provider_name": "Crusoe", "tag": "crusoe/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "min_p", "repetition_penalty", "top_k", "logit_bias", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.94058229352348, "uptime_last_5m": 100, "uptime_last_1d": 99.53826910300236, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@cloudflare", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.293, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000293" } ], "output": [ { "amount": 2.2529999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002253" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 24000, "maxCompletionTokens": 24000, "quantization": "fp8", "status": -2, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 88.69047619047619, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 24000, "pricing": { "prompt": "0.000000293", "completion": "0.000002253", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare/fp8", "quantization": "fp8", "max_completion_tokens": 24000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format", "logprobs", "top_logprobs" ], "status": -2, "uptime_last_30m": 88.69047619047619, "uptime_last_5m": 67.85714285714286, "uptime_last_1d": 89.68004549399899, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@sambanova", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000009" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "SambaNova", "servingProviderSlug": "sambanova", "contextLength": 131072, "maxCompletionTokens": 3072, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 99.06832298136646, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "SambaNova | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000045", "completion": "0.0000009", "discount": 0.25 }, "provider_name": "SambaNova", "tag": "sambanova-turbo", "quantization": "unknown", "max_completion_tokens": 3072, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.06832298136646, "uptime_last_5m": 98.73417721518987, "uptime_last_1d": 97.4389702021086, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@groq", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.59, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000059" } ], "output": [ { "amount": 0.7899999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000079" } ], "cacheRead": [ { "amount": 0.295, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000295" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "Groq", "servingProviderSlug": "groq", "contextLength": 131072, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 99.53246753246752, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Groq | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000059", "completion": "0.00000079", "input_cache_read": "0.000000295", "discount": 0 }, "provider_name": "Groq", "tag": "groq", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.53246753246752, "uptime_last_5m": 100, "uptime_last_1d": 99.91952716027004, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@coreweave", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.71, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000071" } ], "output": [ { "amount": 0.71, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000071" } ], "cacheRead": [ { "amount": 0.71, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000071" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 128000, "maxCompletionTokens": 128000, "quantization": "fp16", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 128000, "pricing": { "prompt": "0.00000071", "completion": "0.00000071", "input_cache_read": "0.00000071", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/fp16", "quantization": "fp16", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 98.67145082857279, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@google", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000072" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000072" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 95.48387096774194, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 128000, "pricing": { "prompt": "0.00000072", "completion": "0.00000072", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "seed", "response_format" ], "status": 0, "uptime_last_30m": 95.48387096774194, "uptime_last_5m": null, "uptime_last_1d": 97.10855427713857, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@google", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000072" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000072" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "Google", "servingProviderSlug": "google", "contextLength": 128000, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Google | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 128000, "pricing": { "prompt": "0.00000072", "completion": "0.00000072", "discount": 0 }, "provider_name": "Google", "tag": "google-vertex/us-central1", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "top_k", "frequency_penalty", "presence_penalty", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.97168101495242, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.3-70b-instruct@together", "name": "Meta: Llama 3.3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000104" } ], "output": [ { "amount": 1.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000104" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.3-70b-instruct", "canonicalSlug": "meta-llama/llama-3.3-70b-instruct", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 131072, "maxCompletionTokens": 2048, "quantization": "unknown", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...", "uptimeLast30m": 98.98989898989899, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | meta-llama/llama-3.3-70b-instruct", "model_id": "meta-llama/llama-3.3-70b-instruct", "model_name": "Meta: Llama 3.3 70B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000104", "completion": "0.00000104", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": 2048, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 98.98989898989899, "uptime_last_5m": null, "uptime_last_1d": 97.9242574804019, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "amazon/nova-lite-v1@amazon-bedrock", "name": "Amazon: Nova Lite 1.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "amazon/nova-lite-v1", "canonicalSlug": "amazon/nova-lite-v1", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 300000, "maxCompletionTokens": 5120, "quantization": "unknown", "status": -5, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "description": "Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...", "uptimeLast30m": 74.01021711366539, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | amazon/nova-lite-v1", "model_id": "amazon/nova-lite-v1", "model_name": "Amazon: Nova Lite 1.0", "context_length": 300000, "pricing": { "prompt": "0.00000006", "completion": "0.00000024", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": 5120, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "status": -5, "uptime_last_30m": 74.01021711366539, "uptime_last_5m": null, "uptime_last_1d": 80.86317376554844, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "amazon/nova-lite-v1@amazon-bedrock", "name": "Amazon: Nova Lite 1.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "amazon/nova-lite-v1", "canonicalSlug": "amazon/nova-lite-v1", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 300000, "maxCompletionTokens": 5120, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "description": "Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | amazon/nova-lite-v1", "model_id": "amazon/nova-lite-v1", "model_name": "Amazon: Nova Lite 1.0", "context_length": 300000, "pricing": { "prompt": "0.00000006", "completion": "0.00000024", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 5120, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9453147943735, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "amazon/nova-micro-v1@amazon-bedrock", "name": "Amazon: Nova Micro 1.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000035" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "amazon/nova-micro-v1", "canonicalSlug": "amazon/nova-micro-v1", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 128000, "maxCompletionTokens": 5120, "quantization": "unknown", "status": -5, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "description": "Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...", "uptimeLast30m": 13.93520055148467, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | amazon/nova-micro-v1", "model_id": "amazon/nova-micro-v1", "model_name": "Amazon: Nova Micro 1.0", "context_length": 128000, "pricing": { "prompt": "0.000000035", "completion": "0.00000014", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": 5120, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "status": -5, "uptime_last_30m": 13.93520055148467, "uptime_last_5m": 50.15197568389058, "uptime_last_1d": 12.287173241355902, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "amazon/nova-micro-v1@amazon-bedrock", "name": "Amazon: Nova Micro 1.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000035" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000014" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "amazon/nova-micro-v1", "canonicalSlug": "amazon/nova-micro-v1", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 128000, "maxCompletionTokens": 5120, "quantization": "unknown", "status": -5, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "description": "Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...", "uptimeLast30m": 65.97084439476502, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | amazon/nova-micro-v1", "model_id": "amazon/nova-micro-v1", "model_name": "Amazon: Nova Micro 1.0", "context_length": 128000, "pricing": { "prompt": "0.000000035", "completion": "0.00000014", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 5120, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "status": -5, "uptime_last_30m": 65.97084439476502, "uptime_last_5m": 95.97420978141217, "uptime_last_1d": 83.4503075338844, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "amazon/nova-pro-v1@amazon-bedrock", "name": "Amazon: Nova Pro 1.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "amazon/nova-pro-v1", "canonicalSlug": "amazon/nova-pro-v1", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 300000, "maxCompletionTokens": 5120, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "description": "Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | amazon/nova-pro-v1", "model_id": "amazon/nova-pro-v1", "model_name": "Amazon: Nova Pro 1.0", "context_length": 300000, "pricing": { "prompt": "0.0000008", "completion": "0.0000032", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock/eu-west-1", "quantization": "unknown", "max_completion_tokens": 5120, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.91673605328893, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "amazon/nova-pro-v1@amazon-bedrock", "name": "Amazon: Nova Pro 1.0", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "amazon/nova-pro-v1", "canonicalSlug": "amazon/nova-pro-v1", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 300000, "maxCompletionTokens": 5120, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Nova", "instruct_type": null }, "description": "Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | amazon/nova-pro-v1", "model_id": "amazon/nova-pro-v1", "model_name": "Amazon: Nova Pro 1.0", "context_length": 300000, "pricing": { "prompt": "0.0000008", "completion": "0.0000032", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 5120, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-2024-11-20@openai", "name": "OpenAI: GPT-4o (2024-11-20)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-2024-11-20", "canonicalSlug": "openai/gpt-4o-2024-11-20", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4o-2024-11-20", "model_id": "openai/gpt-4o-2024-11-20", "model_name": "OpenAI: GPT-4o (2024-11-20)", "context_length": 128000, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "input_cache_read": "0.00000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-large-2407@mistral", "name": "Mistral Large 2407", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-large-2407", "canonicalSlug": "mistralai/mistral-large-2407", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/mistral-large-2407", "model_id": "mistralai/mistral-large-2407", "model_name": "Mistral Large 2407", "context_length": 131072, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-2.5-coder-32b-instruct@cloudflare", "name": "Qwen2.5 Coder 32B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000066" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-2.5-coder-32b-instruct", "canonicalSlug": "qwen/qwen-2.5-coder-32b-instruct", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 32768, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "description": "Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | qwen/qwen-2.5-coder-32b-instruct", "model_id": "qwen/qwen-2.5-coder-32b-instruct", "model_name": "Qwen2.5 Coder 32B Instruct", "context_length": 32768, "pricing": { "prompt": "0.00000066", "completion": "0.000001", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thedrummer/unslopnemo-12b@parasail", "name": "TheDrummer: UnslopNemo 12B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thedrummer/unslopnemo-12b", "canonicalSlug": "thedrummer/unslopnemo-12b", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 1024000, "maxCompletionTokens": 1024000, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "mistral" }, "description": "UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | thedrummer/unslopnemo-12b", "model_id": "thedrummer/unslopnemo-12b", "model_name": "TheDrummer: UnslopNemo 12B", "context_length": 1024000, "pricing": { "prompt": "0.0000004", "completion": "0.0000004", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 1024000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.93126472898665, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thedrummer/unslopnemo-12b@nextbit", "name": "TheDrummer: UnslopNemo 12B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thedrummer/unslopnemo-12b", "canonicalSlug": "thedrummer/unslopnemo-12b", "servingProvider": "NextBit", "servingProviderSlug": "nextbit", "contextLength": 32768, "maxCompletionTokens": 32768, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "repetition_penalty", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "mistral" }, "description": "UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "NextBit | thedrummer/unslopnemo-12b", "model_id": "thedrummer/unslopnemo-12b", "model_name": "TheDrummer: UnslopNemo 12B", "context_length": 32768, "pricing": { "prompt": "0.0000004", "completion": "0.0000004", "discount": 0 }, "provider_name": "NextBit", "tag": "nextbit/fp8", "quantization": "fp8", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "repetition_penalty", "seed", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.36624416277519, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthracite-org/magnum-v4-72b@mancer-2", "name": "Magnum v4 72B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "anthracite-org/magnum-v4-72b", "canonicalSlug": "anthracite-org/magnum-v4-72b", "servingProvider": "Mancer 2", "servingProviderSlug": "mancer-2", "contextLength": 32768, "maxCompletionTokens": 4096, "quantization": "fp8", "status": 0, "supportedParameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "logprobs", "top_logprobs", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "description": "This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mancer 2 | anthracite-org/magnum-v4-72b", "model_id": "anthracite-org/magnum-v4-72b", "model_name": "Magnum v4 72B", "context_length": 32768, "pricing": { "prompt": "0.000003", "completion": "0.000005", "discount": 0 }, "provider_name": "Mancer 2", "tag": "mancer/fp8", "quantization": "fp8", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "logprobs", "top_logprobs", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-2.5-7b-instruct@phala", "name": "Qwen: Qwen2.5 7B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000001" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-2.5-7b-instruct", "canonicalSlug": "qwen/qwen-2.5-7b-instruct", "servingProvider": "Phala", "servingProviderSlug": "phala", "contextLength": 32768, "maxCompletionTokens": 32768, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "description": "Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Phala | qwen/qwen-2.5-7b-instruct", "model_id": "qwen/qwen-2.5-7b-instruct", "model_name": "Qwen: Qwen2.5 7B Instruct", "context_length": 32768, "pricing": { "prompt": "0.0000001", "completion": "0.0000002", "discount": 0 }, "provider_name": "Phala", "tag": "phala", "quantization": "unknown", "max_completion_tokens": 32768, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "min_p", "repetition_penalty", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.96856460886697, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-2.5-7b-instruct@together", "name": "Qwen: Qwen2.5 7B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-2.5-7b-instruct", "canonicalSlug": "qwen/qwen-2.5-7b-instruct", "servingProvider": "Together", "servingProviderSlug": "together", "contextLength": 32768, "maxCompletionTokens": 2048, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "description": "Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...", "uptimeLast30m": 99.72196478220575, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Together | qwen/qwen-2.5-7b-instruct", "model_id": "qwen/qwen-2.5-7b-instruct", "model_name": "Qwen: Qwen2.5 7B Instruct", "context_length": 32768, "pricing": { "prompt": "0.0000003", "completion": "0.0000003", "discount": 0 }, "provider_name": "Together", "tag": "together", "quantization": "unknown", "max_completion_tokens": 2048, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "repetition_penalty", "logit_bias", "min_p", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.72196478220575, "uptime_last_5m": 100, "uptime_last_1d": 98.71919810664068, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "thedrummer/rocinante-12b@parasail", "name": "TheDrummer: Rocinante 12B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "thedrummer/rocinante-12b", "canonicalSlug": "thedrummer/rocinante-12b", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 65536, "maxCompletionTokens": 65536, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "description": "Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | thedrummer/rocinante-12b", "model_id": "thedrummer/rocinante-12b", "model_name": "TheDrummer: Rocinante 12B", "context_length": 65536, "pricing": { "prompt": "0.00000025", "completion": "0.0000005", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 65536, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.2-1b-instruct@cloudflare", "name": "Meta: Llama 3.2 1B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.027, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000027" } ], "output": [ { "amount": 0.201, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000201" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.2-1b-instruct", "canonicalSlug": "meta-llama/llama-3.2-1b-instruct", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 60000, "maxCompletionTokens": 60000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | meta-llama/llama-3.2-1b-instruct", "model_id": "meta-llama/llama-3.2-1b-instruct", "model_name": "Meta: Llama 3.2 1B Instruct", "context_length": 60000, "pricing": { "prompt": "0.000000027", "completion": "0.000000201", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 60000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.2-3b-instruct@parasail", "name": "Meta: Llama 3.2 3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000033" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.2-3b-instruct", "canonicalSlug": "meta-llama/llama-3.2-3b-instruct", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...", "uptimeLast30m": 99.71223021582733, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | meta-llama/llama-3.2-3b-instruct", "model_id": "meta-llama/llama-3.2-3b-instruct", "model_name": "Meta: Llama 3.2 3B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000005", "completion": "0.00000033", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.71223021582733, "uptime_last_5m": null, "uptime_last_1d": 99.84482758620689, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.2-3b-instruct@cloudflare", "name": "Meta: Llama 3.2 3B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.0509, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000509" } ], "output": [ { "amount": 0.335, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000335" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.2-3b-instruct", "canonicalSlug": "meta-llama/llama-3.2-3b-instruct", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 80000, "maxCompletionTokens": 80000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...", "uptimeLast30m": 99.61685823754789, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | meta-llama/llama-3.2-3b-instruct", "model_id": "meta-llama/llama-3.2-3b-instruct", "model_name": "Meta: Llama 3.2 3B Instruct", "context_length": 80000, "pricing": { "prompt": "0.0000000509", "completion": "0.000000335", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare", "quantization": "unknown", "max_completion_tokens": 80000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.61685823754789, "uptime_last_5m": 100, "uptime_last_1d": 97.51136752554638, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-2.5-72b-instruct@deepinfra", "name": "Qwen2.5 72B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000036" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-2.5-72b-instruct", "canonicalSlug": "qwen/qwen-2.5-72b-instruct", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 32768, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "description": "Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...", "uptimeLast30m": 99.91578947368421, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | qwen/qwen-2.5-72b-instruct", "model_id": "qwen/qwen-2.5-72b-instruct", "model_name": "Qwen2.5 72B Instruct", "context_length": 32768, "pricing": { "prompt": "0.00000036", "completion": "0.0000004", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.91578947368421, "uptime_last_5m": 100, "uptime_last_1d": 99.91145554093353, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "qwen/qwen-2.5-72b-instruct@novita", "name": "Qwen2.5 72B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000038" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "qwen/qwen-2.5-72b-instruct", "canonicalSlug": "qwen/qwen-2.5-72b-instruct", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 32000, "maxCompletionTokens": 8192, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Qwen", "instruct_type": "chatml" }, "description": "Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | qwen/qwen-2.5-72b-instruct", "model_id": "qwen/qwen-2.5-72b-instruct", "model_name": "Qwen2.5 72B Instruct", "context_length": 32000, "pricing": { "prompt": "0.00000038", "completion": "0.0000004", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 28.92625143293848, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "cohere/command-r-08-2024@cohere", "name": "Cohere: Command R (08-2024)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "cohere/command-r-08-2024", "canonicalSlug": "cohere/command-r-08-2024", "servingProvider": "Cohere", "servingProviderSlug": "cohere", "contextLength": 128000, "maxCompletionTokens": 4000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Cohere", "instruct_type": null }, "description": "command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cohere | cohere/command-r-08-2024", "model_id": "cohere/command-r-08-2024", "model_name": "Cohere: Command R (08-2024)", "context_length": 128000, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "discount": 0 }, "provider_name": "Cohere", "tag": "cohere", "quantization": "unknown", "max_completion_tokens": 4000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 98.66112181762855, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "cohere/command-r-plus-08-2024@cohere", "name": "Cohere: Command R+ (08-2024)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "cohere/command-r-plus-08-2024", "canonicalSlug": "cohere/command-r-plus-08-2024", "servingProvider": "Cohere", "servingProviderSlug": "cohere", "contextLength": 128000, "maxCompletionTokens": 4000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Cohere", "instruct_type": null }, "description": "command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cohere | cohere/command-r-plus-08-2024", "model_id": "cohere/command-r-plus-08-2024", "model_name": "Cohere: Command R+ (08-2024)", "context_length": 128000, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "discount": 0 }, "provider_name": "Cohere", "tag": "cohere", "quantization": "unknown", "max_completion_tokens": 4000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "top_k", "seed", "structured_outputs", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sao10k/l3.1-euryale-70b@deepinfra", "name": "Sao10K: Llama 3.1 Euryale 70B v2.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000085" } ], "output": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000085" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sao10k/l3.1-euryale-70b", "canonicalSlug": "sao10k/l3.1-euryale-70b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | sao10k/l3.1-euryale-70b", "model_id": "sao10k/l3.1-euryale-70b", "model_name": "Sao10K: Llama 3.1 Euryale 70B v2.2", "context_length": 131072, "pricing": { "prompt": "0.00000085", "completion": "0.00000085", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.78973927670312, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sao10k/l3.1-euryale-70b@novita", "name": "Sao10K: Llama 3.1 Euryale 70B v2.2", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.4504, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014504" } ], "output": [ { "amount": 1.4504, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014504" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sao10k/l3.1-euryale-70b", "canonicalSlug": "sao10k/l3.1-euryale-70b", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 8192, "maxCompletionTokens": 8192, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | sao10k/l3.1-euryale-70b", "model_id": "sao10k/l3.1-euryale-70b", "model_name": "Sao10K: Llama 3.1 Euryale 70B v2.2", "context_length": 8192, "pricing": { "prompt": "0.0000014504", "completion": "0.0000014504", "discount": 0.02 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.61315280464217, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nousresearch/hermes-3-llama-3.1-70b@deepinfra", "name": "Nous: Hermes 3 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000007" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nousresearch/hermes-3-llama-3.1-70b", "canonicalSlug": "nousresearch/hermes-3-llama-3.1-70b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "chatml" }, "description": "Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | nousresearch/hermes-3-llama-3.1-70b", "model_id": "nousresearch/hermes-3-llama-3.1-70b", "model_name": "Nous: Hermes 3 70B Instruct", "context_length": 131072, "pricing": { "prompt": "0.0000007", "completion": "0.0000007", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "nousresearch/hermes-3-llama-3.1-405b@deepinfra", "name": "Nous: Hermes 3 405B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "nousresearch/hermes-3-llama-3.1-405b", "canonicalSlug": "nousresearch/hermes-3-llama-3.1-405b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "chatml" }, "description": "Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | nousresearch/hermes-3-llama-3.1-405b", "model_id": "nousresearch/hermes-3-llama-3.1-405b", "model_name": "Nous: Hermes 3 405B Instruct", "context_length": 131072, "pricing": { "prompt": "0.000001", "completion": "0.000001", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sao10k/l3-lunaris-8b@deepinfra", "name": "Sao10K: Llama 3 8B Lunaris", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sao10k/l3-lunaris-8b", "canonicalSlug": "sao10k/l3-lunaris-8b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 8192, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....", "uptimeLast30m": 99.96590521650187, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | sao10k/l3-lunaris-8b", "model_id": "sao10k/l3-lunaris-8b", "model_name": "Sao10K: Llama 3 8B Lunaris", "context_length": 8192, "pricing": { "prompt": "0.00000004", "completion": "0.00000005", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/turbo", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "status": 0, "uptime_last_30m": 99.96590521650187, "uptime_last_5m": 100, "uptime_last_1d": 99.87102095909415, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sao10k/l3-lunaris-8b@parasail", "name": "Sao10K: Llama 3 8B Lunaris", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sao10k/l3-lunaris-8b", "canonicalSlug": "sao10k/l3-lunaris-8b", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 8192, "maxCompletionTokens": 8192, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....", "uptimeLast30m": 99.86091794158554, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | sao10k/l3-lunaris-8b", "model_id": "sao10k/l3-lunaris-8b", "model_name": "Sao10K: Llama 3 8B Lunaris", "context_length": 8192, "pricing": { "prompt": "0.00000004", "completion": "0.00000005", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/bf16", "quantization": "bf16", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.86091794158554, "uptime_last_5m": 99.15966386554622, "uptime_last_1d": 99.85483320050096, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "sao10k/l3-lunaris-8b@novita", "name": "Sao10K: Llama 3 8B Lunaris", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "sao10k/l3-lunaris-8b", "canonicalSlug": "sao10k/l3-lunaris-8b", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 8192, "maxCompletionTokens": 8192, "quantization": "bf16", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....", "uptimeLast30m": 99.67266775777414, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | sao10k/l3-lunaris-8b", "model_id": "sao10k/l3-lunaris-8b", "model_name": "Sao10K: Llama 3 8B Lunaris", "context_length": 8192, "pricing": { "prompt": "0.00000005", "completion": "0.00000005", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty" ], "status": 0, "uptime_last_30m": 99.67266775777414, "uptime_last_5m": 100, "uptime_last_1d": 99.85712294396431, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-2024-08-06@azure", "name": "OpenAI: GPT-4o (2024-08-06)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-2024-08-06", "canonicalSlug": "openai/gpt-4o-2024-08-06", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4o-2024-08-06", "model_id": "openai/gpt-4o-2024-08-06", "model_name": "OpenAI: GPT-4o (2024-08-06)", "context_length": 128000, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "input_cache_read": "0.00000125", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.96274217585693, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-2024-08-06@openai", "name": "OpenAI: GPT-4o (2024-08-06)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-2024-08-06", "canonicalSlug": "openai/gpt-4o-2024-08-06", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4o-2024-08-06", "model_id": "openai/gpt-4o-2024-08-06", "model_name": "OpenAI: GPT-4o (2024-08-06)", "context_length": 128000, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "input_cache_read": "0.00000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.9855446566856, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.1-70b-instruct@deepinfra", "name": "Meta: Llama 3.1 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.1-70b-instruct", "canonicalSlug": "meta-llama/llama-3.1-70b-instruct", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...", "uptimeLast30m": 99.98339697825004, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | meta-llama/llama-3.1-70b-instruct", "model_id": "meta-llama/llama-3.1-70b-instruct", "model_name": "Meta: Llama 3.1 70B Instruct", "context_length": 131072, "pricing": { "prompt": "0.0000004", "completion": "0.0000004", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/turbo", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "tools", "tool_choice", "logit_bias", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.98339697825004, "uptime_last_5m": 100, "uptime_last_1d": 99.78043351847768, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.1-70b-instruct@amazon-bedrock", "name": "Meta: Llama 3.1 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000072" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000072" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.1-70b-instruct", "canonicalSlug": "meta-llama/llama-3.1-70b-instruct", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 131072, "maxCompletionTokens": 8192, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...", "uptimeLast30m": 99.83588621444201, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | meta-llama/llama-3.1-70b-instruct", "model_id": "meta-llama/llama-3.1-70b-instruct", "model_name": "Meta: Llama 3.1 70B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000072", "completion": "0.00000072", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop" ], "status": 0, "uptime_last_30m": 99.83588621444201, "uptime_last_5m": 100, "uptime_last_1d": 99.59652669225544, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.1-70b-instruct@coreweave", "name": "Meta: Llama 3.1 70B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "cacheRead": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000008" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.1-70b-instruct", "canonicalSlug": "meta-llama/llama-3.1-70b-instruct", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 128000, "maxCompletionTokens": 128000, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "structured_outputs", "logprobs", "top_logprobs", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | meta-llama/llama-3.1-70b-instruct", "model_id": "meta-llama/llama-3.1-70b-instruct", "model_name": "Meta: Llama 3.1 70B Instruct", "context_length": 128000, "pricing": { "prompt": "0.0000008", "completion": "0.0000008", "input_cache_read": "0.0000008", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/bf16", "quantization": "bf16", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "structured_outputs", "logprobs", "top_logprobs", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.97558736792766, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.1-8b-instruct@deepinfra", "name": "Meta: Llama 3.1 8B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.1-8b-instruct", "canonicalSlug": "meta-llama/llama-3.1-8b-instruct", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...", "uptimeLast30m": 99.98861220771333, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | meta-llama/llama-3.1-8b-instruct", "model_id": "meta-llama/llama-3.1-8b-instruct", "model_name": "Meta: Llama 3.1 8B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000002", "completion": "0.00000004", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "status": 0, "uptime_last_30m": 99.98861220771333, "uptime_last_5m": 99.96935335580754, "uptime_last_1d": 99.7617102396514, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.1-8b-instruct@novita", "name": "Meta: Llama 3.1 8B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000002" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.1-8b-instruct", "canonicalSlug": "meta-llama/llama-3.1-8b-instruct", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 16384, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...", "uptimeLast30m": 99.76583842851568, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | meta-llama/llama-3.1-8b-instruct", "model_id": "meta-llama/llama-3.1-8b-instruct", "model_name": "Meta: Llama 3.1 8B Instruct", "context_length": 16384, "pricing": { "prompt": "0.00000002", "completion": "0.00000005", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.76583842851568, "uptime_last_5m": 99.51033732317737, "uptime_last_1d": 98.75305769895198, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.1-8b-instruct@groq", "name": "Meta: Llama 3.1 8B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000005" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.1-8b-instruct", "canonicalSlug": "meta-llama/llama-3.1-8b-instruct", "servingProvider": "Groq", "servingProviderSlug": "groq", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...", "uptimeLast30m": 99.83748050864209, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Groq | meta-llama/llama-3.1-8b-instruct", "model_id": "meta-llama/llama-3.1-8b-instruct", "model_name": "Meta: Llama 3.1 8B Instruct", "context_length": 131072, "pricing": { "prompt": "0.00000005", "completion": "0.00000008", "input_cache_read": "0.000000025", "discount": 0 }, "provider_name": "Groq", "tag": "groq", "quantization": "unknown", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "seed", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.83748050864209, "uptime_last_5m": 99.90533291259072, "uptime_last_1d": 99.9118437076601, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.1-8b-instruct@cloudflare", "name": "Meta: Llama 3.1 8B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15200000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000152" } ], "output": [ { "amount": 0.28700000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000287" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.1-8b-instruct", "canonicalSlug": "meta-llama/llama-3.1-8b-instruct", "servingProvider": "Cloudflare", "servingProviderSlug": "cloudflare", "contextLength": 32000, "maxCompletionTokens": 32000, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Cloudflare | meta-llama/llama-3.1-8b-instruct", "model_id": "meta-llama/llama-3.1-8b-instruct", "model_name": "Meta: Llama 3.1 8B Instruct", "context_length": 32000, "pricing": { "prompt": "0.000000152", "completion": "0.000000287", "discount": 0 }, "provider_name": "Cloudflare", "tag": "cloudflare/fp8", "quantization": "fp8", "max_completion_tokens": 32000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "seed", "repetition_penalty", "frequency_penalty", "presence_penalty", "min_p", "stop", "logit_bias", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 93.79779212001537, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "meta-llama/llama-3.1-8b-instruct@coreweave", "name": "Meta: Llama 3.1 8B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "output": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000022" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "meta-llama/llama-3.1-8b-instruct", "canonicalSlug": "meta-llama/llama-3.1-8b-instruct", "servingProvider": "CoreWeave", "servingProviderSlug": "coreweave", "contextLength": 128000, "maxCompletionTokens": 128000, "quantization": "bf16", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama3", "instruct_type": "llama3" }, "description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "CoreWeave | meta-llama/llama-3.1-8b-instruct", "model_id": "meta-llama/llama-3.1-8b-instruct", "model_name": "Meta: Llama 3.1 8B Instruct", "context_length": 128000, "pricing": { "prompt": "0.00000022", "completion": "0.00000022", "input_cache_read": "0.00000022", "discount": 0 }, "provider_name": "CoreWeave", "tag": "coreweave/bf16", "quantization": "bf16", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "top_k", "repetition_penalty", "frequency_penalty", "presence_penalty", "stop", "seed", "tools", "tool_choice", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.961885402109, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-nemo@deepinfra", "name": "Mistral: Mistral Nemo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.019000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000019" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-nemo", "canonicalSlug": "mistralai/mistral-nemo", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 131072, "maxCompletionTokens": 16384, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "mistral" }, "description": "A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...", "uptimeLast30m": 98.27870460196037, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | mistralai/mistral-nemo", "model_id": "mistralai/mistral-nemo", "model_name": "Mistral: Mistral Nemo", "context_length": 131072, "pricing": { "prompt": "0.000000019", "completion": "0.00000003", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp8", "quantization": "fp8", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "status": 0, "uptime_last_30m": 98.27870460196037, "uptime_last_5m": 98.39309937045668, "uptime_last_1d": 98.71798114133101, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-nemo@parasail", "name": "Mistral: Mistral Nemo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-nemo", "canonicalSlug": "mistralai/mistral-nemo", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 131072, "maxCompletionTokens": 131072, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "mistral" }, "description": "A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...", "uptimeLast30m": 99.9122191011236, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | mistralai/mistral-nemo", "model_id": "mistralai/mistral-nemo", "model_name": "Mistral: Mistral Nemo", "context_length": 131072, "pricing": { "prompt": "0.00000003", "completion": "0.00000003", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp8", "quantization": "fp8", "max_completion_tokens": 131072, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": 99.9122191011236, "uptime_last_5m": 100, "uptime_last_1d": 99.9073063813763, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-nemo@novita", "name": "Mistral: Mistral Nemo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000004" } ], "output": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000017" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-nemo", "canonicalSlug": "mistralai/mistral-nemo", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 60288, "maxCompletionTokens": 16000, "quantization": "fp8", "status": -5, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "mistral" }, "description": "A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...", "uptimeLast30m": 77.2838299951148, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | mistralai/mistral-nemo", "model_id": "mistralai/mistral-nemo", "model_name": "Mistral: Mistral Nemo", "context_length": 60288, "pricing": { "prompt": "0.00000004", "completion": "0.00000017", "discount": 0 }, "provider_name": "Novita", "tag": "novita/fp8", "quantization": "fp8", "max_completion_tokens": 16000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format", "structured_outputs", "logprobs", "top_logprobs" ], "status": -5, "uptime_last_30m": 77.2838299951148, "uptime_last_5m": 90.81632653061224, "uptime_last_1d": 64.49719905011266, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-nemo@io-net", "name": "Mistral: Mistral Nemo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.041999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000042" } ], "output": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000016" } ], "cacheRead": [ { "amount": 0.022, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000022" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-nemo", "canonicalSlug": "mistralai/mistral-nemo", "servingProvider": "Io Net", "servingProviderSlug": "io-net", "contextLength": 128000, "maxCompletionTokens": 128000, "quantization": "fp16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "seed", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "mistral" }, "description": "A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...", "uptimeLast30m": 98.20503330866025, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Io Net | mistralai/mistral-nemo", "model_id": "mistralai/mistral-nemo", "model_name": "Mistral: Mistral Nemo", "context_length": 128000, "pricing": { "prompt": "0.000000042", "completion": "0.00000016", "input_cache_read": "0.000000022", "discount": 0 }, "provider_name": "Io Net", "tag": "io-net/fp16", "quantization": "fp16", "max_completion_tokens": 128000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "frequency_penalty", "presence_penalty", "seed", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 98.20503330866025, "uptime_last_5m": 100, "uptime_last_1d": 95.50237686075836, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-mini@azure", "name": "OpenAI: GPT-4o-mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-mini", "canonicalSlug": "openai/gpt-4o-mini", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...", "uptimeLast30m": 99.95801931111689, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4o-mini", "model_id": "openai/gpt-4o-mini", "model_name": "OpenAI: GPT-4o-mini", "context_length": 128000, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.000000075", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": 99.95801931111689, "uptime_last_5m": 99.96178830722201, "uptime_last_1d": 99.93671813727103, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-mini@openai", "name": "OpenAI: GPT-4o-mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-mini", "canonicalSlug": "openai/gpt-4o-mini", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...", "uptimeLast30m": 99.98531709207946, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4o-mini", "model_id": "openai/gpt-4o-mini", "model_name": "OpenAI: GPT-4o-mini", "context_length": 128000, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.000000075", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.98531709207946, "uptime_last_5m": 99.98573262947639, "uptime_last_1d": 99.9603654151581, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-mini@azure", "name": "OpenAI: GPT-4o-mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.165, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000165" } ], "output": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000066" } ], "cacheRead": [ { "amount": 0.0825, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000825" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-mini", "canonicalSlug": "openai/gpt-4o-mini", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4o-mini", "model_id": "openai/gpt-4o-mini", "model_name": "OpenAI: GPT-4o-mini", "context_length": 128000, "pricing": { "prompt": "0.000000165", "completion": "0.00000066", "input_cache_read": "0.0000000825", "discount": 0 }, "provider_name": "Azure", "tag": "azure/swedencentral", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-mini-2024-07-18@openai", "name": "OpenAI: GPT-4o-mini (2024-07-18)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000015" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-mini-2024-07-18", "canonicalSlug": "openai/gpt-4o-mini-2024-07-18", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4o-mini-2024-07-18", "model_id": "openai/gpt-4o-mini-2024-07-18", "model_name": "OpenAI: GPT-4o-mini (2024-07-18)", "context_length": 128000, "pricing": { "prompt": "0.00000015", "completion": "0.0000006", "input_cache_read": "0.000000075", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-mini:batch@openai", "name": "OpenAI: GPT-4o-mini", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000075" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000000375" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-mini:batch", "canonicalSlug": "openai/gpt-4o-mini", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4o-mini:batch", "model_id": "openai/gpt-4o-mini:batch", "model_name": "OpenAI: GPT-4o-mini", "context_length": 128000, "pricing": { "prompt": "0.000000075", "completion": "0.0000003", "web_search": "0.01", "input_cache_read": "0.0000000375", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "google/gemma-2-27b-it@nextbit", "name": "Google: Gemma 2 27B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000065" } ], "output": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000065" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "google/gemma-2-27b-it", "canonicalSlug": "google/gemma-2-27b-it", "servingProvider": "NextBit", "servingProviderSlug": "nextbit", "contextLength": 8192, "maxCompletionTokens": 2048, "quantization": "int4", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "repetition_penalty", "seed" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Gemini", "instruct_type": "gemma" }, "description": "Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "NextBit | google/gemma-2-27b-it", "model_id": "google/gemma-2-27b-it", "model_name": "Google: Gemma 2 27B", "context_length": 8192, "pricing": { "prompt": "0.00000065", "completion": "0.00000065", "discount": 0 }, "provider_name": "NextBit", "tag": "nextbit/int4", "quantization": "int4", "max_completion_tokens": 2048, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "response_format", "structured_outputs", "repetition_penalty", "seed" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o@azure", "name": "OpenAI: GPT-4o", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o", "canonicalSlug": "openai/gpt-4o", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...", "uptimeLast30m": 99.9297259311314, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4o", "model_id": "openai/gpt-4o", "model_name": "OpenAI: GPT-4o", "context_length": 128000, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.9297259311314, "uptime_last_5m": 100, "uptime_last_1d": 99.95706484011669, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o@openai", "name": "OpenAI: GPT-4o", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o", "canonicalSlug": "openai/gpt-4o", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...", "uptimeLast30m": 99.99008428358948, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4o", "model_id": "openai/gpt-4o", "model_name": "OpenAI: GPT-4o", "context_length": 128000, "pricing": { "prompt": "0.0000025", "completion": "0.00001", "input_cache_read": "0.00000125", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 99.99008428358948, "uptime_last_5m": 100, "uptime_last_1d": 99.94641705457472, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-2024-05-13@azure", "name": "OpenAI: GPT-4o (2024-05-13)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-2024-05-13", "canonicalSlug": "openai/gpt-4o-2024-05-13", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 128000, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4o-2024-05-13", "model_id": "openai/gpt-4o-2024-05-13", "model_name": "OpenAI: GPT-4o (2024-05-13)", "context_length": 128000, "pricing": { "prompt": "0.000005", "completion": "0.000015", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.50519544779812, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o-2024-05-13@openai", "name": "OpenAI: GPT-4o (2024-05-13)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o-2024-05-13", "canonicalSlug": "openai/gpt-4o-2024-05-13", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4o-2024-05-13", "model_id": "openai/gpt-4o-2024-05-13", "model_name": "OpenAI: GPT-4o (2024-05-13)", "context_length": 128000, "pricing": { "prompt": "0.000005", "completion": "0.000015", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 99.89583333333333, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4o:batch@openai", "name": "OpenAI: GPT-4o", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000000625" } ], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4o:batch", "canonicalSlug": "openai/gpt-4o", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 16384, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "architecture": { "modality": "text+image+file->text", "input_modalities": [ "text", "image", "file" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4o:batch", "model_id": "openai/gpt-4o:batch", "model_name": "OpenAI: GPT-4o", "context_length": 128000, "pricing": { "prompt": "0.00000125", "completion": "0.000005", "web_search": "0.01", "input_cache_read": "0.000000625", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "web_search_options", "logit_bias", "logprobs", "top_logprobs", "prediction", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mixtral-8x22b-instruct@mistral", "name": "Mistral: Mixtral 8x22B Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mixtral-8x22b-instruct", "canonicalSlug": "mistralai/mixtral-8x22b-instruct", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 65536, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "mistral" }, "description": "Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/mixtral-8x22b-instruct", "model_id": "mistralai/mixtral-8x22b-instruct", "model_name": "Mistral: Mixtral 8x22B Instruct", "context_length": 65536, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "microsoft/wizardlm-2-8x22b@novita", "name": "WizardLM-2 8x22B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.62, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000062" } ], "output": [ { "amount": 0.62, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000062" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "microsoft/wizardlm-2-8x22b", "canonicalSlug": "microsoft/wizardlm-2-8x22b", "servingProvider": "Novita", "servingProviderSlug": "novita", "contextLength": 65535, "maxCompletionTokens": 8000, "quantization": "bf16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": "vicuna" }, "description": "WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Novita | microsoft/wizardlm-2-8x22b", "model_id": "microsoft/wizardlm-2-8x22b", "model_name": "WizardLM-2 8x22B", "context_length": 65535, "pricing": { "prompt": "0.00000062", "completion": "0.00000062", "discount": 0 }, "provider_name": "Novita", "tag": "novita/bf16", "quantization": "bf16", "max_completion_tokens": 8000, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "response_format" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.98588965711866, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4-turbo@openai", "name": "OpenAI: GPT-4 Turbo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4-turbo", "canonicalSlug": "openai/gpt-4-turbo", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4-turbo", "model_id": "openai/gpt-4-turbo", "model_name": "OpenAI: GPT-4 Turbo", "context_length": 128000, "pricing": { "prompt": "0.00001", "completion": "0.00003", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4-turbo:batch@openai", "name": "OpenAI: GPT-4 Turbo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4-turbo:batch", "canonicalSlug": "openai/gpt-4-turbo", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4-turbo:batch", "model_id": "openai/gpt-4-turbo:batch", "model_name": "OpenAI: GPT-4 Turbo", "context_length": 128000, "pricing": { "prompt": "0.000005", "completion": "0.000015", "web_search": "0.01", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "anthropic/claude-3-haiku@amazon-bedrock", "name": "Anthropic: Claude 3 Haiku", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000003" } ], "cacheWrite": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000003" } ], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "anthropic/claude-3-haiku", "canonicalSlug": "anthropic/claude-3-haiku", "servingProvider": "Amazon Bedrock", "servingProviderSlug": "amazon-bedrock", "contextLength": 200000, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "architecture": { "modality": "text+image->text", "input_modalities": [ "text", "image" ], "output_modalities": [ "text" ], "tokenizer": "Claude", "instruct_type": null }, "description": "Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Amazon Bedrock | anthropic/claude-3-haiku", "model_id": "anthropic/claude-3-haiku", "model_name": "Anthropic: Claude 3 Haiku", "context_length": 200000, "pricing": { "prompt": "0.00000025", "completion": "0.00000125", "web_search": "0.01", "input_cache_read": "0.00000003", "input_cache_write": "0.0000003", "input_cache_write_1h": "0.0000005", "discount": 0 }, "provider_name": "Amazon Bedrock", "tag": "amazon-bedrock", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "top_k", "stop", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.99792835665411, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mistralai/mistral-large@mistral", "name": "Mistral Large", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mistralai/mistral-large", "canonicalSlug": "mistralai/mistral-large", "servingProvider": "Mistral", "servingProviderSlug": "mistral", "contextLength": 128000, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text+file->text", "input_modalities": [ "text", "file" ], "output_modalities": [ "text" ], "tokenizer": "Mistral", "instruct_type": null }, "description": "This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mistral | mistralai/mistral-large", "model_id": "mistralai/mistral-large", "model_name": "Mistral Large", "context_length": 128000, "pricing": { "prompt": "0.000002", "completion": "0.000006", "input_cache_read": "0.0000002", "discount": 0 }, "provider_name": "Mistral", "tag": "mistral", "quantization": "unknown", "max_completion_tokens": null, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.97698292132763, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-3.5-turbo-0613@azure", "name": "OpenAI: GPT-3.5 Turbo (older v0613)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-3.5-turbo-0613", "canonicalSlug": "openai/gpt-3.5-turbo-0613", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 4095, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-3.5-turbo-0613", "model_id": "openai/gpt-3.5-turbo-0613", "model_name": "OpenAI: GPT-3.5 Turbo (older v0613)", "context_length": 4095, "pricing": { "prompt": "0.000001", "completion": "0.000002", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4-turbo-preview@openai", "name": "OpenAI: GPT-4 Turbo Preview", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4-turbo-preview", "canonicalSlug": "openai/gpt-4-turbo-preview", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 128000, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4-turbo-preview", "model_id": "openai/gpt-4-turbo-preview", "model_name": "OpenAI: GPT-4 Turbo Preview", "context_length": 128000, "pricing": { "prompt": "0.00001", "completion": "0.00003", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openrouter/auto", "name": "Auto Router", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": -1000000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "-1" } ], "output": [ { "amount": -1000000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "-1" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "canonicalSlug": "openrouter/auto", "contextLength": 2000000, "architecture": { "modality": "text+image+file+audio+video->text+image", "input_modalities": [ "text", "image", "audio", "file", "video" ], "output_modalities": [ "text", "image" ], "tokenizer": "Router", "instruct_type": null }, "supportedParameters": [ "frequency_penalty", "include_reasoning", "logit_bias", "logprobs", "max_tokens", "min_p", "prediction", "presence_penalty", "reasoning", "reasoning_effort", "repetition_penalty", "response_format", "seed", "stop", "structured_outputs", "temperature", "tool_choice", "tools", "top_a", "top_k", "top_logprobs", "top_p", "web_search_options" ], "description": "Your prompt will be processed by a meta-model and routed to one of dozens of models (see below), optimizing for the best possible output. To see which model was used,...", "endpointCount": 0 } }, { "id": "openai/gpt-3.5-turbo-instruct@openai", "name": "OpenAI: GPT-3.5 Turbo Instruct", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-3.5-turbo-instruct", "canonicalSlug": "openai/gpt-3.5-turbo-instruct", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 4095, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": "chatml" }, "description": "This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-3.5-turbo-instruct", "model_id": "openai/gpt-3.5-turbo-instruct", "model_name": "OpenAI: GPT-3.5 Turbo Instruct", "context_length": 4095, "pricing": { "prompt": "0.0000015", "completion": "0.000002", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-3.5-turbo-16k@openai", "name": "OpenAI: GPT-3.5 Turbo 16k", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-3.5-turbo-16k", "canonicalSlug": "openai/gpt-3.5-turbo-16k", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 16385, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-3.5-turbo-16k", "model_id": "openai/gpt-3.5-turbo-16k", "model_name": "OpenAI: GPT-3.5 Turbo 16k", "context_length": 16385, "pricing": { "prompt": "0.000003", "completion": "0.000004", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-3.5-turbo-16k@azure", "name": "OpenAI: GPT-3.5 Turbo 16k", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-3.5-turbo-16k", "canonicalSlug": "openai/gpt-3.5-turbo-16k", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 16385, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-3.5-turbo-16k", "model_id": "openai/gpt-3.5-turbo-16k", "model_name": "OpenAI: GPT-3.5 Turbo 16k", "context_length": 16385, "pricing": { "prompt": "0.000003", "completion": "0.000004", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "structured_outputs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "mancer/weaver@mancer-2", "name": "Mancer: Weaver (alpha)", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "mancer/weaver", "canonicalSlug": "mancer/weaver", "servingProvider": "Mancer 2", "servingProviderSlug": "mancer-2", "contextLength": 8000, "maxCompletionTokens": 6000, "quantization": "fp8", "status": 0, "supportedParameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "logprobs", "top_logprobs", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama2", "instruct_type": "alpaca" }, "description": "An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mancer 2 | mancer/weaver", "model_id": "mancer/weaver", "model_name": "Mancer: Weaver (alpha)", "context_length": 8000, "pricing": { "prompt": "0.0000005", "completion": "0.00000075", "discount": 0 }, "provider_name": "Mancer 2", "tag": "mancer/fp8", "quantization": "fp8", "max_completion_tokens": 6000, "max_prompt_tokens": null, "supported_parameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "logprobs", "top_logprobs", "structured_outputs" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "undi95/remm-slerp-l2-13b@mancer-2", "name": "ReMM SLERP 13B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000065" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "undi95/remm-slerp-l2-13b", "canonicalSlug": "undi95/remm-slerp-l2-13b", "servingProvider": "Mancer 2", "servingProviderSlug": "mancer-2", "contextLength": 6144, "maxCompletionTokens": 6144, "quantization": "fp8", "status": 0, "supportedParameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "logprobs", "top_logprobs", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama2", "instruct_type": "alpaca" }, "description": "A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mancer 2 | undi95/remm-slerp-l2-13b", "model_id": "undi95/remm-slerp-l2-13b", "model_name": "ReMM SLERP 13B", "context_length": 6144, "pricing": { "prompt": "0.00000045", "completion": "0.00000065", "discount": 0 }, "provider_name": "Mancer 2", "tag": "mancer/fp8", "quantization": "fp8", "max_completion_tokens": 6144, "max_prompt_tokens": null, "supported_parameters": [ "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "logprobs", "top_logprobs", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.983922829582, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "undi95/remm-slerp-l2-13b@nextbit", "name": "ReMM SLERP 13B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000045" } ], "output": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000065" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "undi95/remm-slerp-l2-13b", "canonicalSlug": "undi95/remm-slerp-l2-13b", "servingProvider": "NextBit", "servingProviderSlug": "nextbit", "contextLength": 6144, "maxCompletionTokens": 4096, "quantization": "bf16", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "repetition_penalty", "seed" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama2", "instruct_type": "alpaca" }, "description": "A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "NextBit | undi95/remm-slerp-l2-13b", "model_id": "undi95/remm-slerp-l2-13b", "model_name": "ReMM SLERP 13B", "context_length": 6144, "pricing": { "prompt": "0.00000045", "completion": "0.00000065", "discount": 0 }, "provider_name": "NextBit", "tag": "nextbit/bf16", "quantization": "bf16", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "repetition_penalty", "seed" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 98.94686452848252, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "gryphe/mythomax-l2-13b@nextbit", "name": "MythoMax 13B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "output": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "gryphe/mythomax-l2-13b", "canonicalSlug": "gryphe/mythomax-l2-13b", "servingProvider": "NextBit", "servingProviderSlug": "nextbit", "contextLength": 4096, "maxCompletionTokens": 4096, "quantization": "int4", "status": 0, "supportedParameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "repetition_penalty", "seed" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama2", "instruct_type": "alpaca" }, "description": "One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "NextBit | gryphe/mythomax-l2-13b", "model_id": "gryphe/mythomax-l2-13b", "model_name": "MythoMax 13B", "context_length": 4096, "pricing": { "prompt": "0.00000006", "completion": "0.00000006", "discount": 0 }, "provider_name": "NextBit", "tag": "nextbit/int4", "quantization": "int4", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "repetition_penalty", "seed" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.17473986365268, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "gryphe/mythomax-l2-13b@parasail", "name": "MythoMax 13B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000008" } ], "output": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000011" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "gryphe/mythomax-l2-13b", "canonicalSlug": "gryphe/mythomax-l2-13b", "servingProvider": "Parasail", "servingProviderSlug": "parasail", "contextLength": 4096, "maxCompletionTokens": 4096, "quantization": "fp16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "logprobs", "top_logprobs", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama2", "instruct_type": "alpaca" }, "description": "One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge", "uptimeLast30m": 97.70491803278688, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Parasail | gryphe/mythomax-l2-13b", "model_id": "gryphe/mythomax-l2-13b", "model_name": "MythoMax 13B", "context_length": 4096, "pricing": { "prompt": "0.00000008", "completion": "0.00000011", "discount": 0 }, "provider_name": "Parasail", "tag": "parasail/fp16", "quantization": "fp16", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "frequency_penalty", "presence_penalty", "repetition_penalty", "seed", "stop", "top_k", "logit_bias", "logprobs", "top_logprobs", "structured_outputs" ], "status": 0, "uptime_last_30m": 97.70491803278688, "uptime_last_5m": null, "uptime_last_1d": 97.9464898239195, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "gryphe/mythomax-l2-13b@deepinfra", "name": "MythoMax 13B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "gryphe/mythomax-l2-13b", "canonicalSlug": "gryphe/mythomax-l2-13b", "servingProvider": "DeepInfra", "servingProviderSlug": "deepinfra", "contextLength": 4096, "maxCompletionTokens": 16384, "quantization": "fp16", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama2", "instruct_type": "alpaca" }, "description": "One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "DeepInfra | gryphe/mythomax-l2-13b", "model_id": "gryphe/mythomax-l2-13b", "model_name": "MythoMax 13B", "context_length": 4096, "pricing": { "prompt": "0.0000004", "completion": "0.0000004", "discount": 0 }, "provider_name": "DeepInfra", "tag": "deepinfra/fp16", "quantization": "fp16", "max_completion_tokens": 16384, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "top_k", "seed", "min_p", "response_format", "logit_bias" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": null, "uptime_last_1d": 99.81684981684981, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "gryphe/mythomax-l2-13b@mancer-2", "name": "MythoMax 13B", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000004" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "gryphe/mythomax-l2-13b", "canonicalSlug": "gryphe/mythomax-l2-13b", "servingProvider": "Mancer 2", "servingProviderSlug": "mancer-2", "contextLength": 8192, "maxCompletionTokens": 8192, "quantization": "fp8", "status": 0, "supportedParameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "response_format", "logprobs", "top_logprobs", "structured_outputs" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "Llama2", "instruct_type": "alpaca" }, "description": "One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Mancer 2 | gryphe/mythomax-l2-13b", "model_id": "gryphe/mythomax-l2-13b", "model_name": "MythoMax 13B", "context_length": 8192, "pricing": { "prompt": "0.0000004", "completion": "0.0000006", "discount": 0 }, "provider_name": "Mancer 2", "tag": "mancer/fp8", "quantization": "fp8", "max_completion_tokens": 8192, "max_prompt_tokens": null, "supported_parameters": [ "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "logit_bias", "top_k", "min_p", "seed", "top_a", "response_format", "logprobs", "top_logprobs", "structured_outputs" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 99.71101145989039, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-3.5-turbo@openai", "name": "OpenAI: GPT-3.5 Turbo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000005" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-3.5-turbo", "canonicalSlug": "openai/gpt-3.5-turbo", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 16385, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.", "uptimeLast30m": 100, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-3.5-turbo", "model_id": "openai/gpt-3.5-turbo", "model_name": "OpenAI: GPT-3.5 Turbo", "context_length": 16385, "pricing": { "prompt": "0.0000005", "completion": "0.0000015", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": 100, "uptime_last_5m": 100, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-3.5-turbo:batch@openai", "name": "OpenAI: GPT-3.5 Turbo", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000025" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000075" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.01" } ] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-3.5-turbo:batch", "canonicalSlug": "openai/gpt-3.5-turbo", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 16385, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-3.5-turbo:batch", "model_id": "openai/gpt-3.5-turbo:batch", "model_name": "OpenAI: GPT-3.5 Turbo", "context_length": 16385, "pricing": { "prompt": "0.00000025", "completion": "0.00000075", "web_search": "0.01", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": null, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4@openai", "name": "OpenAI: GPT-4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4", "canonicalSlug": "openai/gpt-4", "servingProvider": "OpenAI", "servingProviderSlug": "openai", "contextLength": 8191, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "OpenAI | openai/gpt-4", "model_id": "openai/gpt-4", "model_name": "OpenAI: GPT-4", "context_length": 8191, "pricing": { "prompt": "0.00003", "completion": "0.00006", "discount": 0 }, "provider_name": "OpenAI", "tag": "openai", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "seed", "max_tokens", "response_format", "structured_outputs", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } }, { "id": "openai/gpt-4@azure", "name": "OpenAI: GPT-4", "provider": "openrouter", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "openrouter", "modelId": "openai/gpt-4", "canonicalSlug": "openai/gpt-4", "servingProvider": "Azure", "servingProviderSlug": "azure", "contextLength": 8191, "maxCompletionTokens": 4096, "quantization": "unknown", "status": 0, "supportedParameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "tools", "tool_choice" ], "architecture": { "modality": "text->text", "input_modalities": [ "text" ], "output_modalities": [ "text" ], "tokenizer": "GPT", "instruct_type": null }, "description": "OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...", "uptimeLast30m": null, "latencyLast30m": null, "throughputLast30m": null, "rawEndpoint": { "name": "Azure | openai/gpt-4", "model_id": "openai/gpt-4", "model_name": "OpenAI: GPT-4", "context_length": 8191, "pricing": { "prompt": "0.00003", "completion": "0.00006", "discount": 0 }, "provider_name": "Azure", "tag": "azure", "quantization": "unknown", "max_completion_tokens": 4096, "max_prompt_tokens": null, "supported_parameters": [ "max_completion_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "logit_bias", "logprobs", "top_logprobs", "response_format", "tools", "tool_choice" ], "status": 0, "uptime_last_30m": null, "uptime_last_5m": null, "uptime_last_1d": 100, "supports_implicit_caching": false, "supports_voice_cloning": false, "latency_last_30m": null, "throughput_last_30m": null } } } ] }, { "provider": "litellm", "source": { "url": "https://api.litellm.ai/model_catalog", "fetchedAt": "2026-08-19T00:56:30.541Z" }, "models": [ { "id": "1024-x-1024/50-steps/bedrock/amazon.nova-canvas-v1:0", "name": "1024-x-1024/50-steps/bedrock/amazon.nova-canvas-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.06, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.06" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 2600, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-09-30", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "1024-x-1024/50-steps/stability.stable-diffusion-xl-v1", "name": "1024-x-1024/50-steps/stability.stable-diffusion-xl-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "1024-x-1024/dall-e-2", "name": "1024-x-1024/dall-e-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 1.9e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.9e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "1024-x-1024/max-steps/stability.stable-diffusion-xl-v1", "name": "1024-x-1024/max-steps/stability.stable-diffusion-xl-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.08, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.08" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "256-x-256/dall-e-2", "name": "256-x-256/dall-e-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 2.4414e-7, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "2.4414e-7" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "512-x-512/50-steps/stability.stable-diffusion-xl-v0", "name": "512-x-512/50-steps/stability.stable-diffusion-xl-v0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.018, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.018" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "512-x-512/dall-e-2", "name": "512-x-512/dall-e-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 6.86e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "6.86e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "512-x-512/max-steps/stability.stable-diffusion-xl-v0", "name": "512-x-512/max-steps/stability.stable-diffusion-xl-v0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.036, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.036" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ai21.j2-mid-v1", "name": "ai21.j2-mid-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8191, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ai21.j2-ultra-v1", "name": "ai21.j2-ultra-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 18.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000188" } ], "output": [ { "amount": 18.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000188" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8191, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ai21.jamba-1-5-large-v1:0", "name": "ai21.jamba-1-5-large-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": "2026-11-26", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ai21.jamba-1-5-mini-v1:0", "name": "ai21.jamba-1-5-mini-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": "2026-11-26", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ai21.jamba-instruct-v1:0", "name": "ai21.jamba-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 70000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "aiml/dall-e-2", "name": "aiml/dall-e-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.026, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.026" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/dall-e-3", "name": "aiml/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.052, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.052" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/flux-pro", "name": "aiml/flux-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.065, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.065" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/flux-pro/v1.1", "name": "aiml/flux-pro/v1.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.052, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.052" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/flux-pro/v1.1-ultra", "name": "aiml/flux-pro/v1.1-ultra", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.063, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.063" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/flux-realism", "name": "aiml/flux-realism", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.046, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.046" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/flux/dev", "name": "aiml/flux/dev", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.033, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.033" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/flux/kontext-max/text-to-image", "name": "aiml/flux/kontext-max/text-to-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.104, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.104" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/flux/kontext-pro/text-to-image", "name": "aiml/flux/kontext-pro/text-to-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.052, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.052" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/flux/schnell", "name": "aiml/flux/schnell", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.004, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.004" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/google/imagen-4.0-ultra-generate-001", "name": "aiml/google/imagen-4.0-ultra-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.078, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.078" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/google/nano-banana-pro", "name": "aiml/google/nano-banana-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.195, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.195" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aiml/openai/gpt-image-2", "name": "aiml/openai/gpt-image-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.054, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.054" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "aiml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.nova-canvas-v1:0", "name": "amazon.nova-canvas-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.06, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.06" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 2600, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-09-30", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.amazon.nova-canvas-v1:0", "name": "us.amazon.nova-canvas-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.06, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.06" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 2600, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-09-30", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.writer.palmyra-x4-v1:0", "name": "us.writer.palmyra-x4-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.writer.palmyra-x5-v1:0", "name": "us.writer.palmyra-x5-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "writer.palmyra-x4-v1:0", "name": "writer.palmyra-x4-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "writer.palmyra-x5-v1:0", "name": "writer.palmyra-x5-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.nova-lite-v1:0", "name": "amazon.nova-lite-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.nova-2-lite-v1:0", "name": "amazon.nova-2-lite-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.nova-2-pro-preview-20251202-v1:0", "name": "amazon.nova-2-pro-preview-20251202-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000021875" }, { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000021875" }, { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000021875" } ], "output": [ { "amount": 17.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000175" } ], "cacheRead": [ { "amount": 0.546875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.46875e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "apac.amazon.nova-2-lite-v1:0", "name": "apac.amazon.nova-2-lite-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "output": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "cacheRead": [ { "amount": 0.0825, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.25e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "apac.amazon.nova-2-pro-preview-20251202-v1:0", "name": "apac.amazon.nova-2-pro-preview-20251202-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000021875" }, { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000021875" }, { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000021875" } ], "output": [ { "amount": 17.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000175" } ], "cacheRead": [ { "amount": 0.546875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.46875e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.amazon.nova-2-lite-v1:0", "name": "eu.amazon.nova-2-lite-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "output": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "cacheRead": [ { "amount": 0.0825, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.25e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.amazon.nova-2-pro-preview-20251202-v1:0", "name": "eu.amazon.nova-2-pro-preview-20251202-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000021875" }, { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000021875" }, { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000021875" } ], "output": [ { "amount": 17.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000175" } ], "cacheRead": [ { "amount": 0.546875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.46875e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.amazon.nova-2-lite-v1:0", "name": "us.amazon.nova-2-lite-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "output": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "cacheRead": [ { "amount": 0.0825, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.25e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.amazon.nova-2-pro-preview-20251202-v1:0", "name": "us.amazon.nova-2-pro-preview-20251202-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000021875" }, { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000021875" }, { "amount": 2.1875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000021875" } ], "output": [ { "amount": 17.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000175" } ], "cacheRead": [ { "amount": 0.546875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.46875e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.nova-2-multimodal-embeddings-v1:0", "name": "amazon.nova-2-multimodal-embeddings-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.35e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00014, "currency": "USD", "units": 1, "pricingType": "second", "modality": "audio", "raw": "0.00014" }, { "amount": 0.0007, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.0007" }, { "amount": 0.00006, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00006" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 8172, "maxOutputTokens": null, "maxTokens": 8172, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.nova-micro-v1:0", "name": "amazon.nova-micro-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-8" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.nova-pro-v1:0", "name": "amazon.nova-pro-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.rerank-v1:0", "name": "amazon.rerank-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.001, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.001" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.titan-embed-image-v1", "name": "amazon.titan-embed-image-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "8e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00006, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00006" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 128, "maxOutputTokens": null, "maxTokens": 128, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.titan-embed-text-v1", "name": "amazon.titan-embed-text-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.titan-embed-g1-text-02", "name": "amazon.titan-embed-g1-text-02", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.titan-embed-text-v2:0", "name": "amazon.titan-embed-text-v2:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.titan-image-generator-v1", "name": "amazon.titan-image-generator-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" }, { "amount": 0.008, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.008" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.titan-image-generator-v2", "name": "amazon.titan-image-generator-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" }, { "amount": 0.008, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.008" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.titan-image-generator-v2:0", "name": "amazon.titan-image-generator-v2:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" }, { "amount": 0.008, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.008" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "twelvelabs.marengo-embed-2-7-v1:0", "name": "twelvelabs.marengo-embed-2-7-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 70, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00007" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": "2026-11-30", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.twelvelabs.marengo-embed-2-7-v1:0", "name": "us.twelvelabs.marengo-embed-2-7-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 70, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00007" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00014, "currency": "USD", "units": 1, "pricingType": "second", "modality": "audio", "raw": "0.00014" }, { "amount": 0.0007, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.0007" }, { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0001" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": "2026-11-30", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "eu.twelvelabs.marengo-embed-2-7-v1:0", "name": "eu.twelvelabs.marengo-embed-2-7-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 70, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.00007" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00014, "currency": "USD", "units": 1, "pricingType": "second", "modality": "audio", "raw": "0.00014" }, { "amount": 0.0007, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.0007" }, { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0001" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": "2026-11-30", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "twelvelabs.pegasus-1-2-v1:0", "name": "twelvelabs.pegasus-1-2-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00049, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.00049" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.twelvelabs.pegasus-1-2-v1:0", "name": "us.twelvelabs.pegasus-1-2-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00049, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.00049" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "eu.twelvelabs.pegasus-1-2-v1:0", "name": "eu.twelvelabs.pegasus-1-2-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00049, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.00049" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.titan-text-express-v1", "name": "amazon.titan-text-express-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000013" } ], "output": [ { "amount": 1.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000017" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 42000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.titan-text-lite-v1", "name": "amazon.titan-text-lite-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 42000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "amazon.titan-text-premier-v1:0", "name": "amazon.titan-text-premier-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 42000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-3-5-haiku-20241022-v1:0", "name": "anthropic.claude-3-5-haiku-20241022-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheWrite": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-haiku-4-5-20251001-v1:0", "name": "anthropic.claude-haiku-4-5-20251001-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-haiku-4-5@20251001", "name": "anthropic.claude-haiku-4-5@20251001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-3-5-sonnet-20240620-v1:0", "name": "anthropic.claude-3-5-sonnet-20240620-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-3-5-sonnet-20241022-v2:0", "name": "anthropic.claude-3-5-sonnet-20241022-v2:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-3-7-sonnet-20240620-v1:0", "name": "anthropic.claude-3-7-sonnet-20240620-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-3-7-sonnet-20250219-v1:0", "name": "anthropic.claude-3-7-sonnet-20250219-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-3-haiku-20240307-v1:0", "name": "anthropic.claude-3-haiku-20240307-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [ { "amount": 0.3125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.125e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-09-10", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-3-opus-20240229-v1:0", "name": "anthropic.claude-3-opus-20240229-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-3-sonnet-20240229-v1:0", "name": "anthropic.claude-3-sonnet-20240229-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-07-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-instant-v1", "name": "anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-opus-4-1-20250805-v1:0", "name": "anthropic.claude-opus-4-1-20250805-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": "2027-01-08", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-opus-4-20250514-v1:0", "name": "anthropic.claude-opus-4-20250514-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-opus-4-5-20251101-v1:0", "name": "anthropic.claude-opus-4-5-20251101-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-opus-4-6-v1", "name": "anthropic.claude-opus-4-6-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "global.anthropic.claude-opus-4-6-v1", "name": "global.anthropic.claude-opus-4-6-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-opus-4-6-v1", "name": "us.anthropic.claude-opus-4-6-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-opus-4-6-v1", "name": "eu.anthropic.claude-opus-4-6-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "au.anthropic.claude-opus-4-6-v1", "name": "au.anthropic.claude-opus-4-6-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-opus-4-7", "name": "anthropic.claude-opus-4-7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-mythos-preview", "name": "anthropic.claude-mythos-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "global.anthropic.claude-opus-4-7", "name": "global.anthropic.claude-opus-4-7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-opus-4-7", "name": "us.anthropic.claude-opus-4-7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-opus-4-7", "name": "eu.anthropic.claude-opus-4-7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "au.anthropic.claude-opus-4-7", "name": "au.anthropic.claude-opus-4-7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-fable-5", "name": "anthropic.claude-fable-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "global.anthropic.claude-fable-5", "name": "global.anthropic.claude-fable-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-fable-5", "name": "us.anthropic.claude-fable-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000055" } ], "cacheRead": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheWrite": [ { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" }, { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000022" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-fable-5", "name": "eu.anthropic.claude-fable-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000055" } ], "cacheRead": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheWrite": [ { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" }, { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000022" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-opus-5", "name": "anthropic.claude-opus-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "global.anthropic.claude-opus-5", "name": "global.anthropic.claude-opus-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-opus-5", "name": "us.anthropic.claude-opus-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-opus-5", "name": "eu.anthropic.claude-opus-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "au.anthropic.claude-opus-5", "name": "au.anthropic.claude-opus-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "jp.anthropic.claude-opus-5", "name": "jp.anthropic.claude-opus-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-opus-4-8", "name": "anthropic.claude-opus-4-8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "global.anthropic.claude-opus-4-8", "name": "global.anthropic.claude-opus-4-8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-opus-4-8", "name": "us.anthropic.claude-opus-4-8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-opus-4-8", "name": "eu.anthropic.claude-opus-4-8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "au.anthropic.claude-opus-4-8", "name": "au.anthropic.claude-opus-4-8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "jp.anthropic.claude-opus-4-8", "name": "jp.anthropic.claude-opus-4-8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "jp.anthropic.claude-opus-4-7", "name": "jp.anthropic.claude-opus-4-7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-sonnet-5", "name": "anthropic.claude-sonnet-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "global.anthropic.claude-sonnet-5", "name": "global.anthropic.claude-sonnet-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-sonnet-5", "name": "us.anthropic.claude-sonnet-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" }, { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-sonnet-5", "name": "eu.anthropic.claude-sonnet-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" }, { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "au.anthropic.claude-sonnet-5", "name": "au.anthropic.claude-sonnet-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" }, { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "jp.anthropic.claude-sonnet-5", "name": "jp.anthropic.claude-sonnet-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" }, { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-sonnet-4-6", "name": "anthropic.claude-sonnet-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "global.anthropic.claude-sonnet-4-6", "name": "global.anthropic.claude-sonnet-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-sonnet-4-6", "name": "us.anthropic.claude-sonnet-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-sonnet-4-6", "name": "eu.anthropic.claude-sonnet-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "au.anthropic.claude-sonnet-4-6", "name": "au.anthropic.claude-sonnet-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "jp.anthropic.claude-sonnet-4-6", "name": "jp.anthropic.claude-sonnet-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-sonnet-4-20250514-v1:0", "name": "anthropic.claude-sonnet-4-20250514-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": "2026-10-14", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "anthropic.claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-v1", "name": "anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anthropic.claude-v2:1", "name": "anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/HuggingFaceH4/zephyr-7b-beta", "name": "anyscale/HuggingFaceH4/zephyr-7b-beta", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/codellama/CodeLlama-34b-Instruct-hf", "name": "anyscale/codellama/CodeLlama-34b-Instruct-hf", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/codellama/CodeLlama-70b-Instruct-hf", "name": "anyscale/codellama/CodeLlama-70b-Instruct-hf", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/google/gemma-7b-it", "name": "anyscale/google/gemma-7b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/meta-llama/Llama-2-13b-chat-hf", "name": "anyscale/meta-llama/Llama-2-13b-chat-hf", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/meta-llama/Llama-2-70b-chat-hf", "name": "anyscale/meta-llama/Llama-2-70b-chat-hf", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/meta-llama/Llama-2-7b-chat-hf", "name": "anyscale/meta-llama/Llama-2-7b-chat-hf", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/meta-llama/Meta-Llama-3-70B-Instruct", "name": "anyscale/meta-llama/Meta-Llama-3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/meta-llama/Meta-Llama-3-8B-Instruct", "name": "anyscale/meta-llama/Meta-Llama-3-8B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/mistralai/Mistral-7B-Instruct-v0.1", "name": "anyscale/mistralai/Mistral-7B-Instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/mistralai/Mixtral-8x22B-Instruct-v0.1", "name": "anyscale/mistralai/Mixtral-8x22B-Instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 65536, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "anyscale/mistralai/Mixtral-8x7B-Instruct-v0.1", "name": "anyscale/mistralai/Mixtral-8x7B-Instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anyscale", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "apac.amazon.nova-lite-v1:0", "name": "apac.amazon.nova-lite-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.063, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.3e-8" } ], "output": [ { "amount": 0.252, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.52e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "apac.amazon.nova-micro-v1:0", "name": "apac.amazon.nova-micro-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.037, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.7e-8" } ], "output": [ { "amount": 0.148, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.48e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "apac.amazon.nova-pro-v1:0", "name": "apac.amazon.nova-pro-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.4e-7" } ], "output": [ { "amount": 3.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000336" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "apac.anthropic.claude-3-5-sonnet-20240620-v1:0", "name": "apac.anthropic.claude-3-5-sonnet-20240620-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-07-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "apac.anthropic.claude-3-5-sonnet-20241022-v2:0", "name": "apac.anthropic.claude-3-5-sonnet-20241022-v2:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-07-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "apac.anthropic.claude-3-haiku-20240307-v1:0", "name": "apac.anthropic.claude-3-haiku-20240307-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [ { "amount": 0.3125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.125e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-09-10", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "apac.anthropic.claude-haiku-4-5-20251001-v1:0", "name": "apac.anthropic.claude-haiku-4-5-20251001-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "apac.anthropic.claude-3-sonnet-20240229-v1:0", "name": "apac.anthropic.claude-3-sonnet-20240229-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-07-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "apac.anthropic.claude-sonnet-4-20250514-v1:0", "name": "apac.anthropic.claude-sonnet-4-20250514-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": "2026-10-14", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "assemblyai/best", "name": "assemblyai/best", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00003333, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00003333" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "assemblyai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "assemblyai/nano", "name": "assemblyai/nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00010278, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00010278" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "assemblyai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "au.anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "au.anthropic.claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" }, { "amount": 24.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002475" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" }, { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" }, { "amount": 8.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000825" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" }, { "amount": 13.200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000132" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/ada", "name": "azure/ada", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "azure", "maxInputTokens": 8191, "maxOutputTokens": null, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/codex-mini", "name": "azure/codex-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-11-15", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/command-r-plus", "name": "azure/command-r-plus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/claude-haiku-4-5", "name": "azure_ai/claude-haiku-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/claude-opus-4-5", "name": "azure_ai/claude-opus-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/claude-opus-4-6", "name": "azure_ai/claude-opus-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/claude-opus-4-7", "name": "azure_ai/claude-opus-4-7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/claude-fable-5", "name": "azure_ai/claude-fable-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/claude-opus-5", "name": "azure_ai/claude-opus-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/claude-opus-4-8", "name": "azure_ai/claude-opus-4-8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/claude-opus-4-1", "name": "azure_ai/claude-opus-4-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/claude-sonnet-4-5", "name": "azure_ai/claude-sonnet-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/claude-sonnet-5", "name": "azure_ai/claude-sonnet-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/claude-sonnet-4-6", "name": "azure_ai/claude-sonnet-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/computer-use-preview", "name": "azure/computer-use-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 8192, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/container", "name": "azure/container", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.03, "currency": "USD", "units": 1, "pricingType": "session", "raw": "0.03" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/gpt-oss-120b", "name": "azure_ai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/gpt-5.5", "name": "azure_ai/gpt-5.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" }, { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure_ai/gpt-5.5-2026-04-23", "name": "azure_ai/gpt-5.5-2026-04-23", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" }, { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure_ai/gpt-5.4", "name": "azure_ai/gpt-5.4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure_ai/gpt-5.4-2026-03-05", "name": "azure_ai/gpt-5.4-2026-03-05", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure_ai/gpt-5.4-pro", "name": "azure_ai/gpt-5.4-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00012" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" }, { "amount": 360, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00036" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00027" }, { "amount": 540, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00054" } ], "cacheRead": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure_ai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": true, "webSearch": true } } }, { "id": "azure_ai/gpt-5.4-pro-2026-03-05", "name": "azure_ai/gpt-5.4-pro-2026-03-05", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00012" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" }, { "amount": 360, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00036" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00027" }, { "amount": 540, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00054" } ], "cacheRead": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure_ai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": true, "webSearch": true } } }, { "id": "azure_ai/gpt-5.4-mini", "name": "azure_ai/gpt-5.4-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" }, { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure_ai/gpt-5.4-mini-2026-03-17", "name": "azure_ai/gpt-5.4-mini-2026-03-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" }, { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure_ai/gpt-5.4-nano", "name": "azure_ai/gpt-5.4-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" }, { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure_ai/gpt-5.4-nano-2026-03-17", "name": "azure_ai/gpt-5.4-nano-2026-03-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" }, { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure_ai/model_router", "name": "azure_ai/model_router", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/eu/gpt-4o-2024-08-06", "name": "azure/eu/gpt-4o-2024-08-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/eu/gpt-4o-2024-11-20", "name": "azure/eu/gpt-4o-2024-11-20", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [], "cacheWrite": [ { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/eu/gpt-4o-mini-2024-07-18", "name": "azure/eu/gpt-4o-mini-2024-07-18", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.165, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.65e-7" } ], "output": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" } ], "cacheRead": [ { "amount": 0.083, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.3e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/eu/gpt-4o-mini-realtime-preview-2024-12-17", "name": "azure/eu/gpt-4o-mini-realtime-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6.6e-7" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000011" } ], "output": [ { "amount": 2.64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000264" }, { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000022" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3.3e-7" } ], "cacheWrite": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3.3e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/eu/gpt-4o-realtime-preview-2024-10-01", "name": "azure/eu/gpt-4o-realtime-preview-2024-10-01", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000055" }, { "amount": 110, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00011" } ], "output": [ { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000022" }, { "amount": 220, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00022" } ], "cacheRead": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000275" } ], "cacheWrite": [ { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000022" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/eu/gpt-4o-realtime-preview-2024-12-17", "name": "azure/eu/gpt-4o-realtime-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000055" }, { "amount": 44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000044" } ], "output": [ { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000022" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000275" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/eu/gpt-5-2025-08-07", "name": "azure/eu/gpt-5-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.1375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.375e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-02-09", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/eu/gpt-5-mini-2025-08-07", "name": "azure/eu/gpt-5-mini-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.0275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-02-09", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/eu/gpt-5.1", "name": "azure/eu/gpt-5.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/eu/gpt-5.1-chat", "name": "azure/eu/gpt-5.1-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-06-29", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/eu/gpt-5.1-codex", "name": "azure/eu/gpt-5.1-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "azure/eu/gpt-5.1-codex-mini", "name": "azure/eu/gpt-5.1-codex-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "azure/eu/gpt-5-nano-2025-08-07", "name": "azure/eu/gpt-5-nano-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-8" } ], "output": [ { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-7" } ], "cacheRead": [ { "amount": 0.0055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-9" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-02-09", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/eu/o1-2024-12-17", "name": "azure/eu/o1-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "output": [ { "amount": 66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000066" } ], "cacheRead": [ { "amount": 8.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000825" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-21", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/eu/o1-mini-2024-09-12", "name": "azure/eu/o1-mini-2024-09-12", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000121" }, { "amount": 0.605, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.05e-7" } ], "output": [ { "amount": 4.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000484" }, { "amount": 2.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000242" } ], "cacheRead": [ { "amount": 0.605, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.05e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/eu/o1-preview-2024-09-12", "name": "azure/eu/o1-preview-2024-09-12", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "output": [ { "amount": 66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000066" } ], "cacheRead": [ { "amount": 8.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000825" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/eu/o3-mini-2025-01-31", "name": "azure/eu/o3-mini-2025-01-31", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000121" }, { "amount": 0.605, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.05e-7" } ], "output": [ { "amount": 4.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000484" }, { "amount": 2.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000242" } ], "cacheRead": [ { "amount": 0.605, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.05e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-01", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/global-standard/gpt-4o-2024-08-06", "name": "azure/global-standard/gpt-4o-2024-08-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/global-standard/gpt-4o-2024-11-20", "name": "azure/global-standard/gpt-4o-2024-11-20", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/global-standard/gpt-4o-mini", "name": "azure/global-standard/gpt-4o-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/global/gpt-4o-2024-08-06", "name": "azure/global/gpt-4o-2024-08-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/global/gpt-4o-2024-11-20", "name": "azure/global/gpt-4o-2024-11-20", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/global/gpt-5.1", "name": "azure/global/gpt-5.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/global/gpt-5.1-chat", "name": "azure/global/gpt-5.1-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-06-29", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/global/gpt-5.1-codex", "name": "azure/global/gpt-5.1-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "azure/global/gpt-5.1-codex-mini", "name": "azure/global/gpt-5.1-codex-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "azure/gpt-3.5-turbo", "name": "azure/gpt-3.5-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 4097, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-3.5-turbo-0125", "name": "azure/gpt-3.5-turbo-0125", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 16384, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2025-03-31", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-3.5-turbo-instruct-0914", "name": "azure/gpt-3.5-turbo-instruct-0914", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "azure_text", "maxInputTokens": 4097, "maxOutputTokens": null, "maxTokens": 4097, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-35-turbo", "name": "azure/gpt-35-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 4097, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-35-turbo-0125", "name": "azure/gpt-35-turbo-0125", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 16384, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2025-05-31", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-35-turbo-1106", "name": "azure/gpt-35-turbo-1106", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 16384, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2025-03-31", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-35-turbo-16k", "name": "azure/gpt-35-turbo-16k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-35-turbo-16k-0613", "name": "azure/gpt-35-turbo-16k-0613", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-35-turbo-instruct", "name": "azure/gpt-35-turbo-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "azure_text", "maxInputTokens": 4097, "maxOutputTokens": null, "maxTokens": 4097, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-35-turbo-instruct-0914", "name": "azure/gpt-35-turbo-instruct-0914", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "azure_text", "maxInputTokens": 4097, "maxOutputTokens": null, "maxTokens": 4097, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4", "name": "azure/gpt-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4-0125-preview", "name": "azure/gpt-4-0125-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4-0613", "name": "azure/gpt-4-0613", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4-1106-preview", "name": "azure/gpt-4-1106-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4-32k", "name": "azure/gpt-4-32k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "output": [ { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 32768, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4-32k-0613", "name": "azure/gpt-4-32k-0613", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "output": [ { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 32768, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4-turbo", "name": "azure/gpt-4-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4-turbo-2024-04-09", "name": "azure/gpt-4-turbo-2024-04-09", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4-turbo-vision-preview", "name": "azure/gpt-4-turbo-vision-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4.1", "name": "azure/gpt-4.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": false } } }, { "id": "azure/gpt-4.1-2025-04-14", "name": "azure/gpt-4.1-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": false } } }, { "id": "azure/gpt-4.1-mini", "name": "azure/gpt-4.1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" }, { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": false } } }, { "id": "azure/gpt-4.1-mini-2025-04-14", "name": "azure/gpt-4.1-mini-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" }, { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": false } } }, { "id": "azure/gpt-4.1-nano", "name": "azure/gpt-4.1-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-4.1-nano-2025-04-14", "name": "azure/gpt-4.1-nano-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-10-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-4.5-preview", "name": "azure/gpt-4.5-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" }, { "amount": 37.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000375" } ], "output": [ { "amount": 150, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00015" }, { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 37.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000375" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-4o", "name": "azure/gpt-4o", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4o-2024-05-13", "name": "azure/gpt-4o-2024-05-13", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-01", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4o-2024-08-06", "name": "azure/gpt-4o-2024-08-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4o-2024-11-20", "name": "azure/gpt-4o-2024-11-20", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-audio-2025-08-28", "name": "azure/gpt-audio-2025-08-28", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00004" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-03-02", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": false, "responseSchema": false, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-audio-1.5-2026-02-23", "name": "azure/gpt-audio-1.5-2026-02-23", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00004" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-08-24", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": false, "responseSchema": false, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-audio-mini-2025-10-06", "name": "azure/gpt-audio-mini-2025-10-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-06", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": false, "responseSchema": false, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-4o-audio-preview-2024-12-17", "name": "azure/gpt-4o-audio-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00004" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": false, "responseSchema": false, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-4o-mini", "name": "azure/gpt-4o-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.165, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.65e-7" } ], "output": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4o-mini-2024-07-18", "name": "azure/gpt-4o-mini-2024-07-18", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.165, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.65e-7" } ], "output": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4o-mini-audio-preview-2024-12-17", "name": "azure/gpt-4o-mini-audio-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00004" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": false, "responseSchema": false, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-4o-mini-realtime-preview-2024-12-17", "name": "azure/gpt-4o-mini-realtime-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-realtime-2025-08-28", "name": "azure/gpt-realtime-2025-08-28", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000032" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000016" }, { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000064" } ], "cacheRead": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" } ], "cacheWrite": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" } ], "other": [ { "amount": 0.000005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000005" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 32000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2027-03-02", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-realtime-1.5-2026-02-23", "name": "azure/gpt-realtime-1.5-2026-02-23", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000032" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000016" }, { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000064" } ], "cacheRead": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" } ], "cacheWrite": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" } ], "other": [ { "amount": 0.000005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000005" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 32000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2027-08-24", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-realtime-mini-2025-10-06", "name": "azure/gpt-realtime-mini-2025-10-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-8" } ], "cacheWrite": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "other": [ { "amount": 8e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "8e-7" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 32000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-4o-mini-transcribe", "name": "azure/gpt-4o-mini-transcribe", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000125" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000125" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "azure", "maxInputTokens": 16000, "maxOutputTokens": 2000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4o-mini-tts", "name": "azure/gpt-4o-mini-tts", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00025, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00025" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4o-realtime-preview-2024-10-01", "name": "azure/gpt-4o-realtime-preview-2024-10-01", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000005" }, { "amount": 100, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0001" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" }, { "amount": 200, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0002" } ], "cacheRead": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "cacheWrite": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-4o-realtime-preview-2024-12-17", "name": "azure/gpt-4o-realtime-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000005" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00004" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-4o-transcribe", "name": "azure/gpt-4o-transcribe", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "azure", "maxInputTokens": 16000, "maxOutputTokens": 2000, "maxTokens": null, "deprecationDate": "2026-10-15", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-4o-transcribe-diarize", "name": "azure/gpt-4o-transcribe-diarize", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "azure", "maxInputTokens": 16000, "maxOutputTokens": 2000, "maxTokens": null, "deprecationDate": "2027-04-15", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-realtime-whisper", "name": "azure/gpt-realtime-whisper", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0002833333333333333, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0002833333333333333" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-5.1-2025-11-13", "name": "azure/gpt-5.1-2025-11-13", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-05-15", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.1-chat-2025-11-13", "name": "azure/gpt-5.1-chat-2025-11-13", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-06-29", "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.1-codex-2025-11-13", "name": "azure/gpt-5.1-codex-2025-11-13", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-05-15", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "azure/gpt-5.1-codex-mini-2025-11-13", "name": "azure/gpt-5.1-codex-mini-2025-11-13", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" }, { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-05-15", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "azure/gpt-5", "name": "azure/gpt-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5-2025-08-07", "name": "azure/gpt-5-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-02-09", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5-chat", "name": "azure/gpt-5-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-06-29", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5-chat-latest", "name": "azure/gpt-5-chat-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-06-29", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5-codex", "name": "azure/gpt-5-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-03-17", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5-mini", "name": "azure/gpt-5-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5-mini-2025-08-07", "name": "azure/gpt-5-mini-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-02-09", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5-nano", "name": "azure/gpt-5-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-9" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5-nano-2025-08-07", "name": "azure/gpt-5-nano-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-9" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-02-09", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5-pro", "name": "azure/gpt-5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-04-07", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.1", "name": "azure/gpt-5.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.1-chat", "name": "azure/gpt-5.1-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-06-29", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.1-codex", "name": "azure/gpt-5.1-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "azure/gpt-5.1-codex-max", "name": "azure/gpt-5.1-codex-max", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-05-18", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "azure/gpt-5.1-codex-mini", "name": "azure/gpt-5.1-codex-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "azure/gpt-5.2", "name": "azure/gpt-5.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.2-2025-12-11", "name": "azure/gpt-5.2-2025-12-11", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" }, { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-06-08", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.2-chat", "name": "azure/gpt-5.2-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" }, { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-06-29", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.2-chat-2025-12-11", "name": "azure/gpt-5.2-chat-2025-12-11", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" }, { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-05-13", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.2-codex", "name": "azure/gpt-5.2-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-07-13", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.3-chat", "name": "azure/gpt-5.3-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" }, { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-06-29", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.3-codex", "name": "azure/gpt-5.3-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-08-24", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.2-pro", "name": "azure/gpt-5.2-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000021" } ], "output": [ { "amount": 168, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000168" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.2-pro-2025-12-11", "name": "azure/gpt-5.2-pro-2025-12-11", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000021" } ], "output": [ { "amount": 168, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000168" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.4", "name": "azure/gpt-5.4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/us/gpt-5.4", "name": "azure/us/gpt-5.4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" }, { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" }, { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" }, { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/eu/gpt-5.4", "name": "azure/eu/gpt-5.4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" }, { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" }, { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" }, { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.4-2026-03-05", "name": "azure/gpt-5.4-2026-03-05", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-09-02", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/us/gpt-5.4-2026-03-05", "name": "azure/us/gpt-5.4-2026-03-05", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" }, { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" }, { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" }, { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-09-02", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/eu/gpt-5.4-2026-03-05", "name": "azure/eu/gpt-5.4-2026-03-05", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" }, { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" }, { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" }, { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-09-02", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/gpt-5.4-pro", "name": "azure/gpt-5.4-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00027" } ], "cacheRead": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.4-pro-2026-03-05", "name": "azure/gpt-5.4-pro-2026-03-05", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00027" } ], "cacheRead": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-09-07", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.6", "name": "azure/gpt-5.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" }, { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.6-sol", "name": "azure/gpt-5.6-sol", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" }, { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2028-01-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.6-terra", "name": "azure/gpt-5.6-terra", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" }, { "amount": 36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000036" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2028-01-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.6-luna", "name": "azure/gpt-5.6-luna", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" }, { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" }, { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" }, { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" }, { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" }, { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2028-01-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/us/gpt-5.6", "name": "azure/us/gpt-5.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" }, { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" }, { "amount": 82.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000825" }, { "amount": 49.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000495" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" }, { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" }, { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/us/gpt-5.6-sol", "name": "azure/us/gpt-5.6-sol", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" }, { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" }, { "amount": 82.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000825" }, { "amount": 49.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000495" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" }, { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" }, { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2028-01-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/us/gpt-5.6-terra", "name": "azure/us/gpt-5.6-terra", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" }, { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" }, { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "output": [ { "amount": 13.200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000132" }, { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" }, { "amount": 19.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000198" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" }, { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-7" }, { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2028-01-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/us/gpt-5.6-luna", "name": "azure/us/gpt-5.6-luna", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" }, { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" }, { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-7" } ], "output": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" }, { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" }, { "amount": 1.9800000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000198" } ], "cacheRead": [ { "amount": 0.022, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-8" }, { "amount": 0.044, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-8" }, { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2028-01-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/eu/gpt-5.6", "name": "azure/eu/gpt-5.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" }, { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" }, { "amount": 82.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000825" }, { "amount": 49.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000495" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" }, { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" }, { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/eu/gpt-5.6-sol", "name": "azure/eu/gpt-5.6-sol", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" }, { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" }, { "amount": 82.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000825" }, { "amount": 49.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000495" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" }, { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" }, { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2028-01-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/eu/gpt-5.6-terra", "name": "azure/eu/gpt-5.6-terra", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" }, { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" }, { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "output": [ { "amount": 13.200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000132" }, { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" }, { "amount": 19.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000198" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" }, { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-7" }, { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2028-01-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/eu/gpt-5.6-luna", "name": "azure/eu/gpt-5.6-luna", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" }, { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" }, { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-7" } ], "output": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" }, { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" }, { "amount": 1.9800000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000198" } ], "cacheRead": [ { "amount": 0.022, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-8" }, { "amount": 0.044, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-8" }, { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2028-01-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.5", "name": "azure/gpt-5.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" }, { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/us/gpt-5.5", "name": "azure/us/gpt-5.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" }, { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" }, { "amount": 82.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000825" }, { "amount": 49.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000495" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" }, { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" }, { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/eu/gpt-5.5", "name": "azure/eu/gpt-5.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" }, { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" }, { "amount": 82.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000825" }, { "amount": 49.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000495" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" }, { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" }, { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.5-2026-04-23", "name": "azure/gpt-5.5-2026-04-23", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" }, { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/us/gpt-5.5-2026-04-23", "name": "azure/us/gpt-5.5-2026-04-23", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" }, { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" }, { "amount": 82.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000825" }, { "amount": 49.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000495" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" }, { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" }, { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/eu/gpt-5.5-2026-04-23", "name": "azure/eu/gpt-5.5-2026-04-23", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" }, { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" }, { "amount": 82.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000825" }, { "amount": 49.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000495" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" }, { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" }, { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.5-pro", "name": "azure/gpt-5.5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00027" } ], "cacheRead": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.5-pro-2026-04-23", "name": "azure/gpt-5.5-pro-2026-04-23", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00027" } ], "cacheRead": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.4-mini", "name": "azure/gpt-5.4-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.4-mini-2026-03-17", "name": "azure/gpt-5.4-mini-2026-03-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-09-21", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.4-nano", "name": "azure/gpt-5.4-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-5.4-nano-2026-03-17", "name": "azure/gpt-5.4-nano-2026-03-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-09-21", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/gpt-image-1", "name": "azure/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" } ], "output": [ { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00004" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-10-23", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/hd/1024-x-1024/dall-e-3", "name": "azure/hd/1024-x-1024/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 7.629e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "7.629e-8" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/hd/1024-x-1792/dall-e-3", "name": "azure/hd/1024-x-1792/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 6.539e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "6.539e-8" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/hd/1792-x-1024/dall-e-3", "name": "azure/hd/1792-x-1024/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 6.539e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "6.539e-8" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/high/1024-x-1024/gpt-image-1", "name": "azure/high/1024-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 1.59263611e-7, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.59263611e-7" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/high/1024-x-1536/gpt-image-1", "name": "azure/high/1024-x-1536/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 1.58945719e-7, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.58945719e-7" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/high/1536-x-1024/gpt-image-1", "name": "azure/high/1536-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 1.58945719e-7, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.58945719e-7" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/low/1024-x-1024/gpt-image-1", "name": "azure/low/1024-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 1.0490417e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.0490417e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/low/1024-x-1536/gpt-image-1", "name": "azure/low/1024-x-1536/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 1.0172526e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.0172526e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/low/1536-x-1024/gpt-image-1", "name": "azure/low/1536-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 1.0172526e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.0172526e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/medium/1024-x-1024/gpt-image-1", "name": "azure/medium/1024-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.0054321e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "4.0054321e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/medium/1024-x-1536/gpt-image-1", "name": "azure/medium/1024-x-1536/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.0054321e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "4.0054321e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/medium/1536-x-1024/gpt-image-1", "name": "azure/medium/1536-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.0054321e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "4.0054321e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-image-1-mini", "name": "azure/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "2e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2027-04-07", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-image-1.5", "name": "azure/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "output": [ { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000032" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-image-1.5-2025-12-16", "name": "azure/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "output": [ { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000032" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2027-06-16", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-image-2", "name": "azure/gpt-image-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00003" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/gpt-image-2-2026-04-21", "name": "azure/gpt-image-2-2026-04-21", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00003" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2027-10-21", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/low/1024-x-1024/gpt-image-1-mini", "name": "azure/low/1024-x-1024/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 2.0751953125e-9, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "2.0751953125e-9" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/low/1024-x-1536/gpt-image-1-mini", "name": "azure/low/1024-x-1536/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 2.0751953125e-9, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "2.0751953125e-9" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/low/1536-x-1024/gpt-image-1-mini", "name": "azure/low/1536-x-1024/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 2.0345052083e-9, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "2.0345052083e-9" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/medium/1024-x-1024/gpt-image-1-mini", "name": "azure/medium/1024-x-1024/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 8.056640625e-9, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "8.056640625e-9" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/medium/1024-x-1536/gpt-image-1-mini", "name": "azure/medium/1024-x-1536/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 8.056640625e-9, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "8.056640625e-9" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/medium/1536-x-1024/gpt-image-1-mini", "name": "azure/medium/1536-x-1024/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 7.9752604167e-9, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "7.9752604167e-9" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/high/1024-x-1024/gpt-image-1-mini", "name": "azure/high/1024-x-1024/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 3.173828125e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "3.173828125e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/high/1024-x-1536/gpt-image-1-mini", "name": "azure/high/1024-x-1536/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 3.173828125e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "3.173828125e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/high/1536-x-1024/gpt-image-1-mini", "name": "azure/high/1536-x-1024/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 3.1575520833e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "3.1575520833e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/mistral-large-2402", "name": "azure/mistral-large-2402", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 32000, "maxOutputTokens": null, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/mistral-large-latest", "name": "azure/mistral-large-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 32000, "maxOutputTokens": null, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o1", "name": "azure/o1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o1-2024-12-17", "name": "azure/o1-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-21", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o1-mini", "name": "azure/o1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000121" } ], "output": [ { "amount": 4.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000484" } ], "cacheRead": [ { "amount": 0.605, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.05e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o1-mini-2024-09-12", "name": "azure/o1-mini-2024-09-12", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o1-preview", "name": "azure/o1-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o1-preview-2024-09-12", "name": "azure/o1-preview-2024-09-12", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o3", "name": "azure/o3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o3-2025-04-16", "name": "azure/o3-2025-04-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-21", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o3-deep-research", "name": "azure/o3-deep-research", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00004" } ], "cacheRead": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-12-26", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "azure/o3-mini", "name": "azure/o3-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o3-mini-2025-01-31", "name": "azure/o3-mini-2025-01-31", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-01", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o3-pro", "name": "azure/o3-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00008" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o3-pro-2025-06-10", "name": "azure/o3-pro-2025-06-10", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00008" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-12-17", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o4-mini", "name": "azure/o4-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/o4-mini-2025-04-16", "name": "azure/o4-mini-2025-04-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-16", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/standard/1024-x-1024/dall-e-2", "name": "azure/standard/1024-x-1024/dall-e-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/standard/1024-x-1024/dall-e-3", "name": "azure/standard/1024-x-1024/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 3.81469e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "3.81469e-8" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/standard/1024-x-1792/dall-e-3", "name": "azure/standard/1024-x-1792/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.359e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "4.359e-8" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/standard/1792-x-1024/dall-e-3", "name": "azure/standard/1792-x-1024/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.359e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "4.359e-8" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/text-embedding-3-large", "name": "azure/text-embedding-3-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.3e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "azure", "maxInputTokens": 8191, "maxOutputTokens": null, "maxTokens": 8191, "deprecationDate": "2028-02-09", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/text-embedding-3-small", "name": "azure/text-embedding-3-small", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "azure", "maxInputTokens": 8191, "maxOutputTokens": null, "maxTokens": 8191, "deprecationDate": "2028-02-09", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/text-embedding-ada-002", "name": "azure/text-embedding-ada-002", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "azure", "maxInputTokens": 8191, "maxOutputTokens": null, "maxTokens": 8191, "deprecationDate": "2028-02-09", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/speech/azure-tts", "name": "azure/speech/azure-tts", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000015, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.000015" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/speech/azure-tts-hd", "name": "azure/speech/azure-tts-hd", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00003, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00003" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/speech/azure-stt", "name": "azure/speech/azure-stt", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0002777778, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0002777778" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/tts-1", "name": "azure/tts-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000015, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.000015" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-12-15", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/tts-1-hd", "name": "azure/tts-1-hd", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00003, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00003" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-12-15", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/us/gpt-4.1-2025-04-14", "name": "azure/us/gpt-4.1-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" }, { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 8.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000088" }, { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": false } } }, { "id": "azure/us/gpt-4.1-mini-2025-04-14", "name": "azure/us/gpt-4.1-mini-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-7" }, { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 1.76, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000176" }, { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": false } } }, { "id": "azure/us/gpt-4.1-nano-2025-04-14", "name": "azure/us/gpt-4.1-nano-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" }, { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-7" }, { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-10-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/us/gpt-4o-2024-08-06", "name": "azure/us/gpt-4o-2024-08-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/us/gpt-4o-2024-11-20", "name": "azure/us/gpt-4o-2024-11-20", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [], "cacheWrite": [ { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/us/gpt-4o-mini-2024-07-18", "name": "azure/us/gpt-4o-mini-2024-07-18", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.165, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.65e-7" } ], "output": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" } ], "cacheRead": [ { "amount": 0.083, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.3e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/us/gpt-4o-mini-realtime-preview-2024-12-17", "name": "azure/us/gpt-4o-mini-realtime-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6.6e-7" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000011" } ], "output": [ { "amount": 2.64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000264" }, { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000022" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3.3e-7" } ], "cacheWrite": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3.3e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/us/gpt-4o-realtime-preview-2024-10-01", "name": "azure/us/gpt-4o-realtime-preview-2024-10-01", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000055" }, { "amount": 110, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00011" } ], "output": [ { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000022" }, { "amount": 220, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00022" } ], "cacheRead": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000275" } ], "cacheWrite": [ { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000022" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/us/gpt-4o-realtime-preview-2024-12-17", "name": "azure/us/gpt-4o-realtime-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000055" }, { "amount": 44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000044" } ], "output": [ { "amount": 22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000022" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000275" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "azure/us/gpt-5-2025-08-07", "name": "azure/us/gpt-5-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.1375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.375e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-02-09", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/us/gpt-5-mini-2025-08-07", "name": "azure/us/gpt-5-mini-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.0275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-02-09", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/us/gpt-5-nano-2025-08-07", "name": "azure/us/gpt-5-nano-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-8" } ], "output": [ { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-7" } ], "cacheRead": [ { "amount": 0.0055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-9" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2027-02-09", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/us/gpt-5.1", "name": "azure/us/gpt-5.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/us/gpt-5.1-chat", "name": "azure/us/gpt-5.1-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-06-29", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "azure/us/gpt-5.1-codex", "name": "azure/us/gpt-5.1-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000138" } ], "output": [ { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "azure/us/gpt-5.1-codex-mini", "name": "azure/us/gpt-5.1-codex-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "azure", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "azure/us/o1-2024-12-17", "name": "azure/us/o1-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "output": [ { "amount": 66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000066" } ], "cacheRead": [ { "amount": 8.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000825" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-21", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/us/o1-mini-2024-09-12", "name": "azure/us/o1-mini-2024-09-12", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000121" }, { "amount": 0.605, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.05e-7" } ], "output": [ { "amount": 4.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000484" }, { "amount": 2.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000242" } ], "cacheRead": [ { "amount": 0.605, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.05e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/us/o1-preview-2024-09-12", "name": "azure/us/o1-preview-2024-09-12", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "output": [ { "amount": 66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000066" } ], "cacheRead": [ { "amount": 8.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000825" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 128000, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/us/o3-2025-04-16", "name": "azure/us/o3-2025-04-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 8.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000088" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-21", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/us/o3-mini-2025-01-31", "name": "azure/us/o3-mini-2025-01-31", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000121" }, { "amount": 0.605, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.05e-7" } ], "output": [ { "amount": 4.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000484" }, { "amount": 2.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000242" } ], "cacheRead": [ { "amount": 0.605, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.05e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-01", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure/us/o4-mini-2025-04-16", "name": "azure/us/o4-mini-2025-04-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000121" } ], "output": [ { "amount": 4.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000484" } ], "cacheRead": [ { "amount": 0.31, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-16", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "azure/whisper-1", "name": "azure/whisper-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" }, { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "azure", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-12-15", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Cohere-embed-v3-english", "name": "azure_ai/Cohere-embed-v3-english", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "azure_ai", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Cohere-embed-v3-multilingual", "name": "azure_ai/Cohere-embed-v3-multilingual", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "azure_ai", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FLUX-1.1-pro", "name": "azure_ai/FLUX-1.1-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FLUX.1-Kontext-pro", "name": "azure_ai/FLUX.1-Kontext-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/flux.2-pro", "name": "azure_ai/flux.2-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-DeepSeek-V3.2", "name": "azure_ai/FW-DeepSeek-V3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.62, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.2e-7" } ], "output": [ { "amount": 1.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000185" } ], "cacheRead": [ { "amount": 0.31, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-DeepSeek-V4-Pro", "name": "azure_ai/FW-DeepSeek-V4-Pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.9250000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001925" } ], "output": [ { "amount": 3.8279999999999994, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003828" } ], "cacheRead": [ { "amount": 0.165, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.65e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 384000, "maxTokens": 384000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-GLM-5", "name": "azure_ai/FW-GLM-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 3.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000352" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-GLM-5.1", "name": "azure_ai/FW-GLM-5.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000154" } ], "output": [ { "amount": 4.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000484" } ], "cacheRead": [ { "amount": 0.286, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.86e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 202800, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-GLM-5.2", "name": "azure_ai/FW-GLM-5.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000154" } ], "output": [ { "amount": 4.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000484" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1048576, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-GLM-5.2-Fast", "name": "azure_ai/FW-GLM-5.2-Fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.0999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000021" } ], "output": [ { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "cacheRead": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1048576, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-Inkling", "name": "azure_ai/FW-Inkling", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 4.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000405" } ], "cacheRead": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.7e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1048576, "maxOutputTokens": 1048576, "maxTokens": 1048576, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-Kimi-K2.5", "name": "azure_ai/FW-Kimi-K2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" } ], "output": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-Kimi-K2.6", "name": "azure_ai/FW-Kimi-K2.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001045" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.176, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.76e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-Kimi-K2.7-Code", "name": "azure_ai/FW-Kimi-K2.7-Code", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.0499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000105" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-Kimi-K3", "name": "azure_ai/FW-Kimi-K3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1048576, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-MiniMax-M2.5", "name": "azure_ai/FW-MiniMax-M2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "output": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "cacheRead": [ { "amount": 0.032999999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-MiniMax-M3", "name": "azure_ai/FW-MiniMax-M3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "output": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" } ], "cacheRead": [ { "amount": 0.06599999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 512000, "maxOutputTokens": 512000, "maxTokens": 512000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/FW-Nemotron-3-Ultra-NVFP4", "name": "azure_ai/FW-Nemotron-3-Ultra-NVFP4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.119, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.19e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/MAI-Image-2.5", "name": "azure_ai/MAI-Image-2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "output": [ { "amount": 47, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000047" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.05, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.05" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/MAI-Image-2.5-Flash", "name": "azure_ai/MAI-Image-2.5-Flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000175" }, { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000175" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000033" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0338, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0338" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/MAI-Image-2e", "name": "azure_ai/MAI-Image-2e", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" } ], "output": [ { "amount": 19.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000195" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.02" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Llama-3.2-11B-Vision-Instruct", "name": "azure_ai/Llama-3.2-11B-Vision-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.37, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.7e-7" } ], "output": [ { "amount": 0.37, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Llama-3.2-90B-Vision-Instruct", "name": "azure_ai/Llama-3.2-90B-Vision-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000204" } ], "output": [ { "amount": 2.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000204" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Llama-3.3-70B-Instruct", "name": "azure_ai/Llama-3.3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.71, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.1e-7" } ], "output": [ { "amount": 0.71, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Llama-4-Maverick-17B-128E-Instruct-FP8", "name": "azure_ai/Llama-4-Maverick-17B-128E-Instruct-FP8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.4100000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000141" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Llama-4-Scout-17B-16E-Instruct", "name": "azure_ai/Llama-4-Scout-17B-16E-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.78, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 10000000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Meta-Llama-3-70B-Instruct", "name": "azure_ai/Meta-Llama-3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 0.37, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 8192, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Meta-Llama-3.1-405B-Instruct", "name": "azure_ai/Meta-Llama-3.1-405B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000533" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Meta-Llama-3.1-70B-Instruct", "name": "azure_ai/Meta-Llama-3.1-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000268" } ], "output": [ { "amount": 3.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000354" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Meta-Llama-3.1-8B-Instruct", "name": "azure_ai/Meta-Llama-3.1-8B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.61, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-3-medium-128k-instruct", "name": "azure_ai/Phi-3-medium-128k-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.7e-7" } ], "output": [ { "amount": 0.6799999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-3-medium-4k-instruct", "name": "azure_ai/Phi-3-medium-4k-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.7e-7" } ], "output": [ { "amount": 0.6799999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-3-mini-128k-instruct", "name": "azure_ai/Phi-3-mini-128k-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-3-mini-4k-instruct", "name": "azure_ai/Phi-3-mini-4k-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-3-small-128k-instruct", "name": "azure_ai/Phi-3-small-128k-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-3-small-8k-instruct", "name": "azure_ai/Phi-3-small-8k-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-3.5-MoE-instruct", "name": "azure_ai/Phi-3.5-MoE-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "output": [ { "amount": 0.64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-3.5-mini-instruct", "name": "azure_ai/Phi-3.5-mini-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-3.5-vision-instruct", "name": "azure_ai/Phi-3.5-vision-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-4", "name": "azure_ai/Phi-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-4-mini-instruct", "name": "azure_ai/Phi-4-mini-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-4-multimodal-instruct", "name": "azure_ai/Phi-4-multimodal-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" } ], "output": [ { "amount": 0.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-4-mini-reasoning", "name": "azure_ai/Phi-4-mini-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/Phi-4-reasoning", "name": "azure_ai/Phi-4-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 32768, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/mistral-document-ai-2505", "name": "azure_ai/mistral-document-ai-2505", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.003, "currency": "USD", "units": 1, "pricingType": "page", "raw": "0.003" } ] } ], "metadata": { "source": "litellm", "mode": "ocr", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/mistral-document-ai-2512", "name": "azure_ai/mistral-document-ai-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.003, "currency": "USD", "units": 1, "pricingType": "page", "raw": "0.003" } ] } ], "metadata": { "source": "litellm", "mode": "ocr", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/doc-intelligence/prebuilt-read", "name": "azure_ai/doc-intelligence/prebuilt-read", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0015, "currency": "USD", "units": 1, "pricingType": "page", "raw": "0.0015" } ] } ], "metadata": { "source": "litellm", "mode": "ocr", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/doc-intelligence/prebuilt-layout", "name": "azure_ai/doc-intelligence/prebuilt-layout", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "page", "raw": "0.01" } ] } ], "metadata": { "source": "litellm", "mode": "ocr", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/doc-intelligence/prebuilt-document", "name": "azure_ai/doc-intelligence/prebuilt-document", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01, "currency": "USD", "units": 1, "pricingType": "page", "raw": "0.01" } ] } ], "metadata": { "source": "litellm", "mode": "ocr", "servingProvider": "azure_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/MAI-DS-R1", "name": "azure_ai/MAI-DS-R1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/cohere-rerank-v3-english", "name": "azure_ai/cohere-rerank-v3-english", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "azure_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/cohere-rerank-v3-multilingual", "name": "azure_ai/cohere-rerank-v3-multilingual", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "azure_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/cohere-rerank-v3.5", "name": "azure_ai/cohere-rerank-v3.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "azure_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/cohere-rerank-v4.0-pro", "name": "azure_ai/cohere-rerank-v4.0-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0025, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.0025" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "azure_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/cohere-rerank-v4.0-fast", "name": "azure_ai/cohere-rerank-v4.0-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "azure_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/deepseek-v3.2", "name": "azure_ai/deepseek-v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.58, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.8e-7" } ], "output": [ { "amount": 1.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000168" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/deepseek-v3.2-speciale", "name": "azure_ai/deepseek-v3.2-speciale", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.58, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.8e-7" } ], "output": [ { "amount": 1.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000168" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/deepseek-r1", "name": "azure_ai/deepseek-r1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/deepseek-v3", "name": "azure_ai/deepseek-v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1400000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000114" } ], "output": [ { "amount": 4.5600000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000456" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/deepseek-v3-0324", "name": "azure_ai/deepseek-v3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1400000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000114" } ], "output": [ { "amount": 4.5600000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000456" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/deepseek-v3.1", "name": "azure_ai/deepseek-v3.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.23, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000123" } ], "output": [ { "amount": 4.94, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000494" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/deepseek-v4-pro", "name": "azure_ai/deepseek-v4-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000174" } ], "output": [ { "amount": 3.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000348" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 384000, "maxTokens": 384000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/deepseek-v4-flash", "name": "azure_ai/deepseek-v4-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.9e-7" } ], "output": [ { "amount": 0.51, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 1000000, "maxOutputTokens": 384000, "maxTokens": 384000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/embed-v-4-0", "name": "azure_ai/embed-v-4-0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/global/grok-3", "name": "azure_ai/global/grok-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "azure_ai/global/grok-3-mini", "name": "azure_ai/global/grok-3-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000127" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "azure_ai/grok-3", "name": "azure_ai/grok-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "azure_ai/grok-3-mini", "name": "azure_ai/grok-3-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000127" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "azure_ai/grok-4", "name": "azure_ai/grok-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "azure_ai/grok-4.3", "name": "azure_ai/grok-4.3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 200000, "maxOutputTokens": 200000, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "azure_ai/grok-4-fast-non-reasoning", "name": "azure_ai/grok-4-fast-non-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "azure_ai/grok-4-fast-reasoning", "name": "azure_ai/grok-4-fast-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "azure_ai/grok-4-1-fast-non-reasoning", "name": "azure_ai/grok-4-1-fast-non-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "azure_ai/grok-4-1-fast-reasoning", "name": "azure_ai/grok-4-1-fast-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "azure_ai/grok-code-fast-1", "name": "azure_ai/grok-code-fast-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "azure_ai/jais-30b-chat", "name": "azure_ai/jais-30b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3200, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0032" } ], "output": [ { "amount": 9710, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00971" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/jamba-instruct", "name": "azure_ai/jamba-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 70000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/kimi-k2.5", "name": "azure_ai/kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/kimi-k2.6", "name": "azure_ai/kimi-k2.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.5e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/ministral-3b", "name": "azure_ai/ministral-3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/mistral-large", "name": "azure_ai/mistral-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/mistral-large-2407", "name": "azure_ai/mistral-large-2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/mistral-large-latest", "name": "azure_ai/mistral-large-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/mistral-large-3", "name": "azure_ai/mistral-large-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 256000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/mistral-medium-2505", "name": "azure_ai/mistral-medium-2505", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/mistral-nemo", "name": "azure_ai/mistral-nemo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 131072, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/mistral-small", "name": "azure_ai/mistral-small", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "azure_ai/mistral-small-2503", "name": "azure_ai/mistral-small-2503", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure_ai", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "babbage-002", "name": "babbage-002", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "text-completion-openai", "maxInputTokens": 16384, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/*/1-month-commitment/cohere.command-light-text-v14", "name": "bedrock/*/1-month-commitment/cohere.command-light-text-v14", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.001902, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.001902" }, { "amount": 0.001902, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.001902" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/*/1-month-commitment/cohere.command-text-v14", "name": "bedrock/*/1-month-commitment/cohere.command-text-v14", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.011, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.011" }, { "amount": 0.011, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.011" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/*/6-month-commitment/cohere.command-light-text-v14", "name": "bedrock/*/6-month-commitment/cohere.command-light-text-v14", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0011416, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0011416" }, { "amount": 0.0011416, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0011416" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/*/6-month-commitment/cohere.command-text-v14", "name": "bedrock/*/6-month-commitment/cohere.command-text-v14", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0066027, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0066027" }, { "amount": 0.0066027, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0066027" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-instant-v1", "name": "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01475, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.01475" }, { "amount": 0.01475, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.01475" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v1", "name": "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0455, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0455" }, { "amount": 0.0455, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0455" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2:1", "name": "bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0455, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0455" }, { "amount": 0.0455, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0455" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-instant-v1", "name": "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.008194, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.008194" }, { "amount": 0.008194, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.008194" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v1", "name": "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02527, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.02527" }, { "amount": 0.02527, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.02527" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2:1", "name": "bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02527, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.02527" }, { "amount": 0.02527, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.02527" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/anthropic.claude-instant-v1", "name": "bedrock/ap-northeast-1/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.23, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000223" } ], "output": [ { "amount": 7.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000755" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/anthropic.claude-v1", "name": "bedrock/ap-northeast-1/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/anthropic.claude-v2:1", "name": "bedrock/ap-northeast-1/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/deepseek.v3.2", "name": "bedrock/ap-northeast-1/deepseek.v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.4e-7" } ], "output": [ { "amount": 2.2199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000222" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/minimax.minimax-m2.1", "name": "bedrock/ap-northeast-1/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/minimax.minimax-m2.5", "name": "bedrock/ap-northeast-1/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/moonshotai.kimi-k2-thinking", "name": "bedrock/ap-northeast-1/moonshotai.kimi-k2-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.73, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.3e-7" } ], "output": [ { "amount": 3.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000303" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/moonshotai.kimi-k2.5", "name": "bedrock/ap-northeast-1/moonshotai.kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-northeast-1/qwen.qwen3-coder-next", "name": "bedrock/ap-northeast-1/qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/moonshotai.kimi-k2-thinking", "name": "bedrock/moonshotai.kimi-k2-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.73, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.3e-7" } ], "output": [ { "amount": 3.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000303" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/moonshotai.kimi-k2.5", "name": "bedrock/moonshotai.kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000303" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-south-1/meta.llama3-70b-instruct-v1:0", "name": "bedrock/ap-south-1/meta.llama3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000318" } ], "output": [ { "amount": 4.199999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000042" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-south-1/meta.llama3-8b-instruct-v1:0", "name": "bedrock/ap-south-1/meta.llama3-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-south-1/deepseek.v3.2", "name": "bedrock/ap-south-1/deepseek.v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.4e-7" } ], "output": [ { "amount": 2.2199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000222" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-south-1/minimax.minimax-m2.1", "name": "bedrock/ap-south-1/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-south-1/minimax.minimax-m2.5", "name": "bedrock/ap-south-1/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-south-1/moonshotai.kimi-k2-thinking", "name": "bedrock/ap-south-1/moonshotai.kimi-k2-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.71, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.1e-7" } ], "output": [ { "amount": 2.94, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000294" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-south-1/moonshotai.kimi-k2.5", "name": "bedrock/ap-south-1/moonshotai.kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-south-1/qwen.qwen3-coder-next", "name": "bedrock/ap-south-1/qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-southeast-2/minimax.minimax-m2.5", "name": "bedrock/ap-southeast-2/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.309, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.09e-7" } ], "output": [ { "amount": 1.236, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001236" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-southeast-3/deepseek.v3.2", "name": "bedrock/ap-southeast-3/deepseek.v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.4e-7" } ], "output": [ { "amount": 2.2199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000222" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ap-southeast-3/minimax.minimax-m2.1", "name": "bedrock/ap-southeast-3/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-southeast-3/minimax.minimax-m2.5", "name": "bedrock/ap-southeast-3/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-southeast-3/moonshotai.kimi-k2.5", "name": "bedrock/ap-southeast-3/moonshotai.kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ap-southeast-3/qwen.qwen3-coder-next", "name": "bedrock/ap-southeast-3/qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/ca-central-1/meta.llama3-70b-instruct-v1:0", "name": "bedrock/ca-central-1/meta.llama3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000305" } ], "output": [ { "amount": 4.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000403" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/ca-central-1/meta.llama3-8b-instruct-v1:0", "name": "bedrock/ca-central-1/meta.llama3-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 0.69, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-north-1/deepseek.v3.2", "name": "bedrock/eu-north-1/deepseek.v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.4e-7" } ], "output": [ { "amount": 2.2199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000222" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-north-1/minimax.minimax-m2.1", "name": "bedrock/eu-north-1/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-north-1/minimax.minimax-m2.5", "name": "bedrock/eu-north-1/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-north-1/moonshotai.kimi-k2.5", "name": "bedrock/eu-north-1/moonshotai.kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1", "name": "bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.01635, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.01635" }, { "amount": 0.01635, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.01635" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-central-1/1-month-commitment/anthropic.claude-v1", "name": "bedrock/eu-central-1/1-month-commitment/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0415, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0415" }, { "amount": 0.0415, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0415" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2:1", "name": "bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0415, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0415" }, { "amount": 0.0415, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0415" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-central-1/6-month-commitment/anthropic.claude-instant-v1", "name": "bedrock/eu-central-1/6-month-commitment/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.009083, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.009083" }, { "amount": 0.009083, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.009083" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-central-1/6-month-commitment/anthropic.claude-v1", "name": "bedrock/eu-central-1/6-month-commitment/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02305, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.02305" }, { "amount": 0.02305, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.02305" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2:1", "name": "bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02305, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.02305" }, { "amount": 0.02305, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.02305" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-central-1/anthropic.claude-instant-v1", "name": "bedrock/eu-central-1/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000248" } ], "output": [ { "amount": 8.379999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000838" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-central-1/anthropic.claude-v1", "name": "bedrock/eu-central-1/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-central-1/anthropic.claude-v2:1", "name": "bedrock/eu-central-1/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-central-1/minimax.minimax-m2.1", "name": "bedrock/eu-central-1/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-central-1/minimax.minimax-m2.5", "name": "bedrock/eu-central-1/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-central-1/qwen.qwen3-coder-next", "name": "bedrock/eu-central-1/qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-west-1/meta.llama3-70b-instruct-v1:0", "name": "bedrock/eu-west-1/meta.llama3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.8600000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000286" } ], "output": [ { "amount": 3.78, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000378" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-west-1/meta.llama3-8b-instruct-v1:0", "name": "bedrock/eu-west-1/meta.llama3-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.2e-7" } ], "output": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-west-1/minimax.minimax-m2.1", "name": "bedrock/eu-west-1/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-west-1/minimax.minimax-m2.5", "name": "bedrock/eu-west-1/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-west-1/qwen.qwen3-coder-next", "name": "bedrock/eu-west-1/qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-west-2/meta.llama3-70b-instruct-v1:0", "name": "bedrock/eu-west-2/meta.llama3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000345" } ], "output": [ { "amount": 4.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000455" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-west-2/meta.llama3-8b-instruct-v1:0", "name": "bedrock/eu-west-2/meta.llama3-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.9e-7" } ], "output": [ { "amount": 0.78, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-west-2/minimax.minimax-m2.1", "name": "bedrock/eu-west-2/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.47, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.7e-7" } ], "output": [ { "amount": 1.8599999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000186" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-west-2/minimax.minimax-m2.5", "name": "bedrock/eu-west-2/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.47, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.7e-7" } ], "output": [ { "amount": 1.8599999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000186" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-west-2/qwen.qwen3-coder-next", "name": "bedrock/eu-west-2/qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.78, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.8e-7" } ], "output": [ { "amount": 1.8599999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000186" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-west-3/mistral.mistral-7b-instruct-v0:2", "name": "bedrock/eu-west-3/mistral.mistral-7b-instruct-v0:2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-west-3/mistral.mistral-large-2402-v1:0", "name": "bedrock/eu-west-3/mistral.mistral-large-2402-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000104" } ], "output": [ { "amount": 31.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000312" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-west-3/mistral.mixtral-8x7b-instruct-v0:1", "name": "bedrock/eu-west-3/mistral.mixtral-8x7b-instruct-v0:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.59, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.9e-7" } ], "output": [ { "amount": 0.9099999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/eu-south-1/minimax.minimax-m2.1", "name": "bedrock/eu-south-1/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-south-1/minimax.minimax-m2.5", "name": "bedrock/eu-south-1/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/eu-south-1/qwen.qwen3-coder-next", "name": "bedrock/eu-south-1/qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/invoke/anthropic.claude-3-5-sonnet-20240620-v1:0", "name": "bedrock/invoke/anthropic.claude-3-5-sonnet-20240620-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/sa-east-1/meta.llama3-70b-instruct-v1:0", "name": "bedrock/sa-east-1/meta.llama3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 4.45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000445" } ], "output": [ { "amount": 5.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000588" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/sa-east-1/meta.llama3-8b-instruct-v1:0", "name": "bedrock/sa-east-1/meta.llama3-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000101" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/sa-east-1/deepseek.v3.2", "name": "bedrock/sa-east-1/deepseek.v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.4e-7" } ], "output": [ { "amount": 2.2199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000222" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/sa-east-1/minimax.minimax-m2.1", "name": "bedrock/sa-east-1/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/sa-east-1/minimax.minimax-m2.5", "name": "bedrock/sa-east-1/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/sa-east-1/moonshotai.kimi-k2-thinking", "name": "bedrock/sa-east-1/moonshotai.kimi-k2-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.73, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.3e-7" } ], "output": [ { "amount": 3.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000303" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/sa-east-1/moonshotai.kimi-k2.5", "name": "bedrock/sa-east-1/moonshotai.kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/sa-east-1/qwen.qwen3-coder-next", "name": "bedrock/sa-east-1/qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000144" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1", "name": "bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.011, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.011" }, { "amount": 0.011, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.011" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/1-month-commitment/anthropic.claude-v1", "name": "bedrock/us-east-1/1-month-commitment/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0175, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0175" }, { "amount": 0.0175, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0175" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/1-month-commitment/anthropic.claude-v2:1", "name": "bedrock/us-east-1/1-month-commitment/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0175, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0175" }, { "amount": 0.0175, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0175" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/6-month-commitment/anthropic.claude-instant-v1", "name": "bedrock/us-east-1/6-month-commitment/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00611, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00611" }, { "amount": 0.00611, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00611" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/6-month-commitment/anthropic.claude-v1", "name": "bedrock/us-east-1/6-month-commitment/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00972, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00972" }, { "amount": 0.00972, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00972" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/6-month-commitment/anthropic.claude-v2:1", "name": "bedrock/us-east-1/6-month-commitment/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00972, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00972" }, { "amount": 0.00972, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00972" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/anthropic.claude-instant-v1", "name": "bedrock/us-east-1/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/anthropic.claude-v1", "name": "bedrock/us-east-1/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/anthropic.claude-v2:1", "name": "bedrock/us-east-1/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/meta.llama3-70b-instruct-v1:0", "name": "bedrock/us-east-1/meta.llama3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000265" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/meta.llama3-8b-instruct-v1:0", "name": "bedrock/us-east-1/meta.llama3-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/mistral.mistral-7b-instruct-v0:2", "name": "bedrock/us-east-1/mistral.mistral-7b-instruct-v0:2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/mistral.mistral-large-2402-v1:0", "name": "bedrock/us-east-1/mistral.mistral-large-2402-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/mistral.mixtral-8x7b-instruct-v0:1", "name": "bedrock/us-east-1/mistral.mixtral-8x7b-instruct-v0:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/deepseek.v3.2", "name": "bedrock/us-east-1/deepseek.v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.62, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.2e-7" } ], "output": [ { "amount": 1.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000185" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/minimax.minimax-m2.1", "name": "bedrock/us-east-1/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-east-1/minimax.minimax-m2.5", "name": "bedrock/us-east-1/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-east-1/moonshotai.kimi-k2-thinking", "name": "bedrock/us-east-1/moonshotai.kimi-k2-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/moonshotai.kimi-k2.5", "name": "bedrock/us-east-1/moonshotai.kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-east-1/qwen.qwen3-coder-next", "name": "bedrock/us-east-1/qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-east-2/deepseek.v3.2", "name": "bedrock/us-east-2/deepseek.v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.62, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.2e-7" } ], "output": [ { "amount": 1.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000185" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-2/minimax.minimax-m2.1", "name": "bedrock/us-east-2/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-east-2/minimax.minimax-m2.5", "name": "bedrock/us-east-2/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-east-2/moonshotai.kimi-k2-thinking", "name": "bedrock/us-east-2/moonshotai.kimi-k2-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-2/moonshotai.kimi-k2.5", "name": "bedrock/us-east-2/moonshotai.kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-east-2/qwen.qwen3-coder-next", "name": "bedrock/us-east-2/qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/amazon.nova-pro-v1:0", "name": "bedrock/us-gov-east-1/amazon.nova-pro-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.96, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.6e-7" } ], "output": [ { "amount": 3.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000384" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/amazon.titan-embed-text-v1", "name": "bedrock/us-gov-east-1/amazon.titan-embed-text-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/amazon.titan-embed-text-v2:0", "name": "bedrock/us-gov-east-1/amazon.titan-embed-text-v2:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/amazon.titan-text-express-v1", "name": "bedrock/us-gov-east-1/amazon.titan-text-express-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000013" } ], "output": [ { "amount": 1.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000017" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 42000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/amazon.titan-text-lite-v1", "name": "bedrock/us-gov-east-1/amazon.titan-text-lite-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 42000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/amazon.titan-text-premier-v1:0", "name": "bedrock/us-gov-east-1/amazon.titan-text-premier-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 42000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/anthropic.claude-3-5-sonnet-20240620-v1:0", "name": "bedrock/us-gov-east-1/anthropic.claude-3-5-sonnet-20240620-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-07-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/anthropic.claude-3-haiku-20240307-v1:0", "name": "bedrock/us-gov-east-1/anthropic.claude-3-haiku-20240307-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-09-10", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "bedrock/us-gov-east-1/anthropic.claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" }, { "amount": 7.199999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000072" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/claude-sonnet-4-5-20250929-v1:0", "name": "bedrock/us-gov-east-1/claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" }, { "amount": 7.199999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000072" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/meta.llama3-70b-instruct-v1:0", "name": "bedrock/us-gov-east-1/meta.llama3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000265" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/meta.llama3-8b-instruct-v1:0", "name": "bedrock/us-gov-east-1/meta.llama3-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 2.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000265" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/amazon.nova-pro-v1:0", "name": "bedrock/us-gov-west-1/amazon.nova-pro-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.96, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.6e-7" } ], "output": [ { "amount": 3.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000384" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/amazon.titan-embed-text-v1", "name": "bedrock/us-gov-west-1/amazon.titan-embed-text-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/amazon.titan-embed-text-v2:0", "name": "bedrock/us-gov-west-1/amazon.titan-embed-text-v2:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/amazon.titan-text-express-v1", "name": "bedrock/us-gov-west-1/amazon.titan-text-express-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000013" } ], "output": [ { "amount": 1.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000017" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 42000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/amazon.titan-text-lite-v1", "name": "bedrock/us-gov-west-1/amazon.titan-text-lite-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 42000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/amazon.titan-text-premier-v1:0", "name": "bedrock/us-gov-west-1/amazon.titan-text-premier-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 42000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/anthropic.claude-3-7-sonnet-20250219-v1:0", "name": "bedrock/us-gov-west-1/anthropic.claude-3-7-sonnet-20250219-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-07-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/anthropic.claude-3-5-sonnet-20240620-v1:0", "name": "bedrock/us-gov-west-1/anthropic.claude-3-5-sonnet-20240620-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-07-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/anthropic.claude-3-haiku-20240307-v1:0", "name": "bedrock/us-gov-west-1/anthropic.claude-3-haiku-20240307-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-09-10", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "bedrock/us-gov-west-1/anthropic.claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" }, { "amount": 7.199999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000072" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/claude-sonnet-4-5-20250929-v1:0", "name": "bedrock/us-gov-west-1/claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" }, { "amount": 7.199999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000072" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/meta.llama3-70b-instruct-v1:0", "name": "bedrock/us-gov-west-1/meta.llama3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000265" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/meta.llama3-8b-instruct-v1:0", "name": "bedrock/us-gov-west-1/meta.llama3-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 2.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000265" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-1/meta.llama3-70b-instruct-v1:0", "name": "bedrock/us-west-1/meta.llama3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000265" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-1/meta.llama3-8b-instruct-v1:0", "name": "bedrock/us-west-1/meta.llama3-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/1-month-commitment/anthropic.claude-instant-v1", "name": "bedrock/us-west-2/1-month-commitment/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.011, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.011" }, { "amount": 0.011, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.011" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/1-month-commitment/anthropic.claude-v1", "name": "bedrock/us-west-2/1-month-commitment/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0175, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0175" }, { "amount": 0.0175, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0175" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/1-month-commitment/anthropic.claude-v2:1", "name": "bedrock/us-west-2/1-month-commitment/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0175, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0175" }, { "amount": 0.0175, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0175" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/6-month-commitment/anthropic.claude-instant-v1", "name": "bedrock/us-west-2/6-month-commitment/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00611, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00611" }, { "amount": 0.00611, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00611" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/6-month-commitment/anthropic.claude-v1", "name": "bedrock/us-west-2/6-month-commitment/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00972, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00972" }, { "amount": 0.00972, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00972" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/6-month-commitment/anthropic.claude-v2:1", "name": "bedrock/us-west-2/6-month-commitment/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00972, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00972" }, { "amount": 0.00972, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00972" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/anthropic.claude-instant-v1", "name": "bedrock/us-west-2/anthropic.claude-instant-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/anthropic.claude-v1", "name": "bedrock/us-west-2/anthropic.claude-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/anthropic.claude-v2:1", "name": "bedrock/us-west-2/anthropic.claude-v2:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 100000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/mistral.mistral-7b-instruct-v0:2", "name": "bedrock/us-west-2/mistral.mistral-7b-instruct-v0:2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/mistral.mistral-large-2402-v1:0", "name": "bedrock/us-west-2/mistral.mistral-large-2402-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/mistral.mixtral-8x7b-instruct-v0:1", "name": "bedrock/us-west-2/mistral.mixtral-8x7b-instruct-v0:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/deepseek.v3.2", "name": "bedrock/us-west-2/deepseek.v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.62, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.2e-7" } ], "output": [ { "amount": 1.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000185" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/minimax.minimax-m2.1", "name": "bedrock/us-west-2/minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-west-2/minimax.minimax-m2.5", "name": "bedrock/us-west-2/minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-west-2/moonshotai.kimi-k2-thinking", "name": "bedrock/us-west-2/moonshotai.kimi-k2-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-west-2/moonshotai.kimi-k2.5", "name": "bedrock/us-west-2/moonshotai.kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-west-2/qwen.qwen3-coder-next", "name": "bedrock/us-west-2/qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us.anthropic.claude-3-5-haiku-20241022-v1:0", "name": "bedrock/us.anthropic.claude-3-5-haiku-20241022-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheWrite": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "black_forest_labs/flux-kontext-pro", "name": "black_forest_labs/flux-kontext-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "black_forest_labs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "black_forest_labs/flux-kontext-max", "name": "black_forest_labs/flux-kontext-max", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.08, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.08" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "black_forest_labs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "black_forest_labs/flux-pro-1.0-fill", "name": "black_forest_labs/flux-pro-1.0-fill", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.05, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.05" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "black_forest_labs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "black_forest_labs/flux-pro-1.0-expand", "name": "black_forest_labs/flux-pro-1.0-expand", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.05, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.05" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "black_forest_labs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "black_forest_labs/flux-pro-1.1", "name": "black_forest_labs/flux-pro-1.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "black_forest_labs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "black_forest_labs/flux-pro-1.1-ultra", "name": "black_forest_labs/flux-pro-1.1-ultra", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.06, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.06" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "black_forest_labs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "black_forest_labs/flux-dev", "name": "black_forest_labs/flux-dev", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.025, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.025" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "black_forest_labs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "black_forest_labs/flux-pro", "name": "black_forest_labs/flux-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.05, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.05" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "black_forest_labs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cerebras/llama-3.3-70b", "name": "cerebras/llama-3.3-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.5e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cerebras", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cerebras/llama3.1-70b", "name": "cerebras/llama3.1-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cerebras", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cerebras/llama3.1-8b", "name": "cerebras/llama3.1-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cerebras", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cerebras/gpt-oss-120b", "name": "cerebras/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cerebras", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "cerebras/qwen-3-32b", "name": "cerebras/qwen-3-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cerebras", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cerebras/zai-glm-4.6", "name": "cerebras/zai-glm-4.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "output": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cerebras", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cerebras/zai-glm-4.7", "name": "cerebras/zai-glm-4.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "output": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cerebras", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "chatdolphin", "name": "chatdolphin", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nlp_cloud", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "chatgpt-4o-latest", "name": "chatgpt-4o-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-02-17", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-transcribe-diarize", "name": "gpt-4o-transcribe-diarize", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "openai", "maxInputTokens": 16000, "maxOutputTokens": 2000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "claude-haiku-4-5-20251001", "name": "claude-haiku-4-5-20251001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-haiku-4-5", "name": "claude-haiku-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-3-7-sonnet-20250219", "name": "claude-3-7-sonnet-20250219", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": "2026-02-19", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "claude-3-haiku-20240307", "name": "claude-3-haiku-20240307", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-04-20", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-3-opus-20240229", "name": "claude-3-opus-20240229", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-01-05", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-4-opus-20250514", "name": "claude-4-opus-20250514", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": "2026-06-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-4-sonnet-20250514", "name": "claude-4-sonnet-20250514", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": "2026-06-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "claude-sonnet-4-5", "name": "claude-sonnet-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-sonnet-4-5-20250929", "name": "claude-sonnet-4-5-20250929", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "claude-sonnet-5", "name": "claude-sonnet-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-sonnet-4-6", "name": "claude-sonnet-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-sonnet-4-5-20250929-v1:0", "name": "claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-opus-4-1", "name": "claude-opus-4-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": "2026-08-05", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-opus-4-1-20250805", "name": "claude-opus-4-1-20250805", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": "2026-08-05", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-opus-4-20250514", "name": "claude-opus-4-20250514", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": "2026-06-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-opus-4-5-20251101", "name": "claude-opus-4-5-20251101", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-opus-4-5", "name": "claude-opus-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-opus-4-6", "name": "claude-opus-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-opus-4-6-20260205", "name": "claude-opus-4-6-20260205", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-opus-4-7", "name": "claude-opus-4-7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-opus-4-7-20260416", "name": "claude-opus-4-7-20260416", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-fable-5", "name": "claude-fable-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-opus-5", "name": "claude-opus-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-opus-4-8", "name": "claude-opus-4-8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-sonnet-4-20250514", "name": "claude-sonnet-4-20250514", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": "2026-06-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/meta/llama-2-7b-chat-fp16", "name": "cloudflare/@cf/meta/llama-2-7b-chat-fp16", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.923, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001923" } ], "output": [ { "amount": 1.923, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001923" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 3072, "maxOutputTokens": 3072, "maxTokens": 3072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/meta/llama-2-7b-chat-int8", "name": "cloudflare/@cf/meta/llama-2-7b-chat-int8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.923, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001923" } ], "output": [ { "amount": 1.923, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001923" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 2048, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/mistral/mistral-7b-instruct-v0.1", "name": "cloudflare/@cf/mistral/mistral-7b-instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.923, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001923" } ], "output": [ { "amount": 1.923, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001923" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@hf/thebloke/codellama-7b-instruct-awq", "name": "cloudflare/@hf/thebloke/codellama-7b-instruct-awq", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.923, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001923" } ], "output": [ { "amount": 1.923, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001923" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/openai/gpt-oss-120b", "name": "cloudflare/@cf/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/google/gemma-2b-it-lora", "name": "cloudflare/@cf/google/gemma-2b-it-lora", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/meta/llama-3.2-3b-instruct", "name": "cloudflare/@cf/meta/llama-3.2-3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.0509, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.09e-8" } ], "output": [ { "amount": 0.335, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.35e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 80000, "maxOutputTokens": 80000, "maxTokens": 80000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/meta/llama-guard-3-8b", "name": "cloudflare/@cf/meta/llama-guard-3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.48400000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.84e-7" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/mistral/mistral-7b-instruct-v0.2-lora", "name": "cloudflare/@cf/mistral/mistral-7b-instruct-v0.2-lora", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 15000, "maxOutputTokens": 15000, "maxTokens": 15000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/moonshotai/kimi-k2.7-code", "name": "cloudflare/@cf/moonshotai/kimi-k2.7-code", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.5e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.9e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b", "name": "cloudflare/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.49699999999999994, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.97e-7" } ], "output": [ { "amount": 4.881, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004881" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 80000, "maxOutputTokens": 80000, "maxTokens": 80000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/meta/llama-3.1-8b-instruct-fp8", "name": "cloudflare/@cf/meta/llama-3.1-8b-instruct-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15200000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.52e-7" } ], "output": [ { "amount": 0.28700000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.87e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/meta/llama-3.2-1b-instruct", "name": "cloudflare/@cf/meta/llama-3.2-1b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.027, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-8" } ], "output": [ { "amount": 0.201, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.01e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 60000, "maxOutputTokens": 60000, "maxTokens": 60000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/moonshotai/kimi-k2.6", "name": "cloudflare/@cf/moonshotai/kimi-k2.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.5e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/zai-org/glm-4.7-flash", "name": "cloudflare/@cf/zai-org/glm-4.7-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.060500000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.05e-8" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/meta-llama/llama-2-7b-chat-hf-lora", "name": "cloudflare/@cf/meta-llama/llama-2-7b-chat-hf-lora", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/meta/llama-3.3-70b-instruct-fp8-fast", "name": "cloudflare/@cf/meta/llama-3.3-70b-instruct-fp8-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.293, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.93e-7" } ], "output": [ { "amount": 2.2529999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002253" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 24000, "maxOutputTokens": 24000, "maxTokens": 24000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/ibm-granite/granite-4.0-h-micro", "name": "cloudflare/@cf/ibm-granite/granite-4.0-h-micro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.017, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.7e-8" } ], "output": [ { "amount": 0.112, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.12e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 131000, "maxOutputTokens": 131000, "maxTokens": 131000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct", "name": "cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/zai-org/glm-5.2", "name": "cloudflare/@cf/zai-org/glm-5.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/nvidia/nemotron-3-120b-a12b", "name": "cloudflare/@cf/nvidia/nemotron-3-120b-a12b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/aisingapore/gemma-sea-lion-v4-27b-it", "name": "cloudflare/@cf/aisingapore/gemma-sea-lion-v4-27b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.351, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.51e-7" } ], "output": [ { "amount": 0.5549999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.55e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/qwen/qwen3-30b-a3b-fp8", "name": "cloudflare/@cf/qwen/qwen3-30b-a3b-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.0509, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.09e-8" } ], "output": [ { "amount": 0.335, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.35e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/google/gemma-7b-it-lora", "name": "cloudflare/@cf/google/gemma-7b-it-lora", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 3500, "maxOutputTokens": 3500, "maxTokens": 3500, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/google/gemma-4-26b-a4b-it", "name": "cloudflare/@cf/google/gemma-4-26b-a4b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/mistralai/mistral-small-3.1-24b-instruct", "name": "cloudflare/@cf/mistralai/mistral-small-3.1-24b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.351, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.51e-7" } ], "output": [ { "amount": 0.5549999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.55e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/meta/llama-3.2-11b-vision-instruct", "name": "cloudflare/@cf/meta/llama-3.2-11b-vision-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.048499999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.85e-8" } ], "output": [ { "amount": 0.6759999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.76e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/openai/gpt-oss-20b", "name": "cloudflare/@cf/openai/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/meta/llama-4-scout-17b-16e-instruct", "name": "cloudflare/@cf/meta/llama-4-scout-17b-16e-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 131000, "maxOutputTokens": 131000, "maxTokens": 131000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cloudflare/@cf/qwen/qwq-32b", "name": "cloudflare/@cf/qwen/qwq-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cloudflare", "maxInputTokens": 24000, "maxOutputTokens": 24000, "maxTokens": 24000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "codestral/codestral-2405", "name": "codestral/codestral-2405", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "codestral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "codestral/codestral-latest", "name": "codestral/codestral-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "codestral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "codex-mini-latest", "name": "codex-mini-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-02-12", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "cohere.command-light-text-v14", "name": "cohere.command-light-text-v14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cohere.command-r-plus-v1:0", "name": "cohere.command-r-plus-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-08-19", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cohere.command-r-v1:0", "name": "cohere.command-r-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-08-19", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cohere.command-text-v14", "name": "cohere.command-text-v14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cohere.embed-english-v3", "name": "cohere.embed-english-v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cohere.embed-multilingual-v3", "name": "cohere.embed-multilingual-v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cohere.embed-v4:0", "name": "cohere.embed-v4:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cohere/embed-v4.0", "name": "cohere/embed-v4.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "cohere", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "cohere.rerank-v3-5:0", "name": "cohere.rerank-v3-5:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "command", "name": "command", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "cohere", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "command-a-03-2025", "name": "command-a-03-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cohere_chat", "maxInputTokens": 256000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "command-light", "name": "command-light", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cohere_chat", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "command-nightly", "name": "command-nightly", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "cohere", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "command-r", "name": "command-r", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cohere_chat", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "command-r-08-2024", "name": "command-r-08-2024", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cohere_chat", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "command-r-plus", "name": "command-r-plus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cohere_chat", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "command-r-plus-08-2024", "name": "command-r-plus-08-2024", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cohere_chat", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "command-r7b-12-2024", "name": "command-r7b-12-2024", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-8" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "cohere_chat", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "computer-use-preview", "name": "computer-use-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "azure", "maxInputTokens": 8192, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "dall-e-2", "name": "dall-e-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.02" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-05-12", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dall-e-3", "name": "dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-05-12", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepseek-chat", "name": "deepseek-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "output": [ { "amount": 0.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.2e-7" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "deepseek-reasoner", "name": "deepseek-reasoner", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "output": [ { "amount": 0.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.2e-7" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "dashscope/deepseek-v4-flash", "name": "dashscope/deepseek-v4-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 1000000, "maxOutputTokens": 393216, "maxTokens": 393216, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/deepseek-v4-flash-0731", "name": "dashscope/deepseek-v4-flash-0731", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 1000000, "maxOutputTokens": 393216, "maxTokens": 393216, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/deepseek-v4-pro", "name": "dashscope/deepseek-v4-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "output": [ { "amount": 4.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000048" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 1000000, "maxOutputTokens": 393216, "maxTokens": 393216, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/glm-5.1", "name": "dashscope/glm-5.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 202745, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/glm-5.2", "name": "dashscope/glm-5.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 1048576, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/kimi-k2.7-code", "name": "dashscope/kimi-k2.7-code", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.5e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.9e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 229376, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen-coder", "name": "dashscope/qwen-coder", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 1000000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen-max", "name": "dashscope/qwen-max", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "output": [ { "amount": 6.3999999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000064" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 30720, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen-plus", "name": "dashscope/qwen-plus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 129024, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen-plus-2025-01-25", "name": "dashscope/qwen-plus-2025-01-25", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 129024, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen-plus-2025-04-28", "name": "dashscope/qwen-plus-2025-04-28", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 129024, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen-plus-2025-07-14", "name": "dashscope/qwen-plus-2025-07-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 129024, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen-turbo", "name": "dashscope/qwen-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 129024, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen-turbo-2024-11-01", "name": "dashscope/qwen-turbo-2024-11-01", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen-turbo-2025-04-28", "name": "dashscope/qwen-turbo-2025-04-28", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 1000000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen-turbo-latest", "name": "dashscope/qwen-turbo-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 1000000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen3-next-80b-a3b-instruct", "name": "dashscope/qwen3-next-80b-a3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen3-next-80b-a3b-thinking", "name": "dashscope/qwen3-next-80b-a3b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen3-vl-235b-a22b-instruct", "name": "dashscope/qwen3-vl-235b-a22b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen3-vl-235b-a22b-thinking", "name": "dashscope/qwen3-vl-235b-a22b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen3-vl-32b-instruct", "name": "dashscope/qwen3-vl-32b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "output": [ { "amount": 0.64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen3-vl-32b-thinking", "name": "dashscope/qwen3-vl-32b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "output": [ { "amount": 2.87, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000287" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen3.7-max", "name": "dashscope/qwen3.7-max", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 991808, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwen3.8-max", "name": "dashscope/qwen3.8-max", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 991808, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "dashscope/qwq-plus", "name": "dashscope/qwq-plus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "dashscope", "maxInputTokens": 98304, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-bge-large-en", "name": "databricks/databricks-bge-large-en", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.10003000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.0003e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "databricks", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-claude-3-7-sonnet", "name": "databricks/databricks-claude-3-7-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.9999900000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000029999900000000002" } ], "output": [ { "amount": 15.000020000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015000020000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-claude-haiku-4-5", "name": "databricks/databricks-claude-haiku-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.0000200000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000100002" } ], "output": [ { "amount": 5.00003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000500003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-claude-opus-4", "name": "databricks/databricks-claude-opus-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15.000020000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015000020000000002" } ], "output": [ { "amount": 75.00003000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00007500003000000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-claude-opus-4-1", "name": "databricks/databricks-claude-opus-4-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15.000020000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015000020000000002" } ], "output": [ { "amount": 75.00003000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00007500003000000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-claude-opus-4-5", "name": "databricks/databricks-claude-opus-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.00003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000500003" } ], "output": [ { "amount": 25.000010000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025000010000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-claude-sonnet-4", "name": "databricks/databricks-claude-sonnet-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.9999900000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000029999900000000002" } ], "output": [ { "amount": 15.000020000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015000020000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-claude-sonnet-4-1", "name": "databricks/databricks-claude-sonnet-4-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.9999900000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000029999900000000002" } ], "output": [ { "amount": 15.000020000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015000020000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-claude-sonnet-4-5", "name": "databricks/databricks-claude-sonnet-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.9999900000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000029999900000000002" } ], "output": [ { "amount": 15.000020000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015000020000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-gemini-2-5-flash", "name": "databricks/databricks-gemini-2-5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.30001999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.0001999999999996e-7" } ], "output": [ { "amount": 2.49998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000249998" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-gemini-2-5-pro", "name": "databricks/databricks-gemini-2-5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.24999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000124999" } ], "output": [ { "amount": 9.999990000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009999990000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-gemma-3-12b", "name": "databricks/databricks-gemma-3-12b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15000999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5000999999999998e-7" } ], "output": [ { "amount": 0.5000100000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.0001e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 128000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-gpt-5", "name": "databricks/databricks-gpt-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.24999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000124999" } ], "output": [ { "amount": 9.999990000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009999990000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-gpt-5-1", "name": "databricks/databricks-gpt-5-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.24999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000124999" } ], "output": [ { "amount": 9.999990000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009999990000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-gpt-5-mini", "name": "databricks/databricks-gpt-5-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.24997000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4997000000000006e-7" } ], "output": [ { "amount": 1.9999700000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000019999700000000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-gpt-5-nano", "name": "databricks/databricks-gpt-5-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049980000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.998e-8" } ], "output": [ { "amount": 0.39998000000000006, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.9998000000000007e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-gpt-oss-120b", "name": "databricks/databricks-gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15000999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5000999999999998e-7" } ], "output": [ { "amount": 0.59997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.9997e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-gpt-oss-20b", "name": "databricks/databricks-gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.30001999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.0001999999999996e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-gte-large-en", "name": "databricks/databricks-gte-large-en", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12999000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.2999000000000001e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "databricks", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-llama-2-70b-chat", "name": "databricks/databricks-llama-2-70b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5000100000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.0001e-7" } ], "output": [ { "amount": 1.5000300000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015000300000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-llama-4-maverick", "name": "databricks/databricks-llama-4-maverick", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5000100000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.0001e-7" } ], "output": [ { "amount": 1.5000300000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015000300000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-meta-llama-3-1-405b-instruct", "name": "databricks/databricks-meta-llama-3-1-405b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.00003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000500003" } ], "output": [ { "amount": 15.000020000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015000020000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-meta-llama-3-1-8b-instruct", "name": "databricks/databricks-meta-llama-3-1-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15000999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5000999999999998e-7" } ], "output": [ { "amount": 0.45003000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5003000000000007e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-meta-llama-3-3-70b-instruct", "name": "databricks/databricks-meta-llama-3-3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5000100000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.0001e-7" } ], "output": [ { "amount": 1.5000300000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015000300000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-meta-llama-3-70b-instruct", "name": "databricks/databricks-meta-llama-3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.0000200000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000100002" } ], "output": [ { "amount": 2.9999900000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000029999900000000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-mixtral-8x7b-instruct", "name": "databricks/databricks-mixtral-8x7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5000100000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.0001e-7" } ], "output": [ { "amount": 1.0000200000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000100002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-mpt-30b-instruct", "name": "databricks/databricks-mpt-30b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.0000200000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000100002" } ], "output": [ { "amount": 1.0000200000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000100002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "databricks/databricks-mpt-7b-instruct", "name": "databricks/databricks-mpt-7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5000100000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.0001e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "databricks", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dataforseo/search", "name": "dataforseo/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.003, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.003" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "dataforseo", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "davinci-002", "name": "davinci-002", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "text-completion-openai", "maxInputTokens": 16384, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/base", "name": "deepgram/base", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00020833, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00020833" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/base-conversationalai", "name": "deepgram/base-conversationalai", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00020833, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00020833" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/base-finance", "name": "deepgram/base-finance", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00020833, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00020833" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/base-general", "name": "deepgram/base-general", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00020833, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00020833" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/base-meeting", "name": "deepgram/base-meeting", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00020833, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00020833" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/base-phonecall", "name": "deepgram/base-phonecall", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00020833, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00020833" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/base-video", "name": "deepgram/base-video", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00020833, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00020833" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/base-voicemail", "name": "deepgram/base-voicemail", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00020833, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00020833" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/enhanced", "name": "deepgram/enhanced", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00024167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00024167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/enhanced-finance", "name": "deepgram/enhanced-finance", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00024167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00024167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/enhanced-general", "name": "deepgram/enhanced-general", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00024167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00024167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/enhanced-meeting", "name": "deepgram/enhanced-meeting", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00024167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00024167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/enhanced-phonecall", "name": "deepgram/enhanced-phonecall", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00024167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00024167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova", "name": "deepgram/nova", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-2", "name": "deepgram/nova-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-2-atc", "name": "deepgram/nova-2-atc", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-2-automotive", "name": "deepgram/nova-2-automotive", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-2-conversationalai", "name": "deepgram/nova-2-conversationalai", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-2-drivethru", "name": "deepgram/nova-2-drivethru", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-2-finance", "name": "deepgram/nova-2-finance", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-2-general", "name": "deepgram/nova-2-general", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-2-meeting", "name": "deepgram/nova-2-meeting", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-2-phonecall", "name": "deepgram/nova-2-phonecall", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-2-video", "name": "deepgram/nova-2-video", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-2-voicemail", "name": "deepgram/nova-2-voicemail", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-3", "name": "deepgram/nova-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-3-general", "name": "deepgram/nova-3-general", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-3-medical", "name": "deepgram/nova-3-medical", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00008667, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00008667" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-general", "name": "deepgram/nova-general", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/nova-phonecall", "name": "deepgram/nova-phonecall", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00007167, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00007167" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/whisper", "name": "deepgram/whisper", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/whisper-base", "name": "deepgram/whisper-base", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/whisper-large", "name": "deepgram/whisper-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/whisper-medium", "name": "deepgram/whisper-medium", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/whisper-small", "name": "deepgram/whisper-small", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepgram/whisper-tiny", "name": "deepgram/whisper-tiny", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "deepgram", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Gryphe/MythoMax-L2-13b", "name": "deepinfra/Gryphe/MythoMax-L2-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/NousResearch/Hermes-3-Llama-3.1-405B", "name": "deepinfra/NousResearch/Hermes-3-Llama-3.1-405B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/NousResearch/Hermes-3-Llama-3.1-70B", "name": "deepinfra/NousResearch/Hermes-3-Llama-3.1-70B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/QwQ-32B", "name": "deepinfra/Qwen/QwQ-32B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen2.5-72B-Instruct", "name": "deepinfra/Qwen/Qwen2.5-72B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.39, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen2.5-7B-Instruct", "name": "deepinfra/Qwen/Qwen2.5-7B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen2.5-VL-32B-Instruct", "name": "deepinfra/Qwen/Qwen2.5-VL-32B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen3-14B", "name": "deepinfra/Qwen/Qwen3-14B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen3-235B-A22B", "name": "deepinfra/Qwen/Qwen3-235B-A22B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "output": [ { "amount": 0.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "deepinfra/Qwen/Qwen3-235B-A22B-Instruct-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "deepinfra/Qwen/Qwen3-235B-A22B-Thinking-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 2.9000000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000029" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen3-30B-A3B", "name": "deepinfra/Qwen/Qwen3-30B-A3B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen3-32B", "name": "deepinfra/Qwen/Qwen3-32B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct", "name": "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo", "name": "deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.9e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen3-Next-80B-A3B-Instruct", "name": "deepinfra/Qwen/Qwen3-Next-80B-A3B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Qwen/Qwen3-Next-80B-A3B-Thinking", "name": "deepinfra/Qwen/Qwen3-Next-80B-A3B-Thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo", "name": "deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Sao10K/L3.1-70B-Euryale-v2.2", "name": "deepinfra/Sao10K/L3.1-70B-Euryale-v2.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/Sao10K/L3.3-70B-Euryale-v2.3", "name": "deepinfra/Sao10K/L3.3-70B-Euryale-v2.3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/allenai/olmOCR-7B-0725-FP8", "name": "deepinfra/allenai/olmOCR-7B-0725-FP8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/anthropic/claude-3-7-sonnet-latest", "name": "deepinfra/anthropic/claude-3-7-sonnet-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 200000, "maxOutputTokens": 200000, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/anthropic/claude-4-opus", "name": "deepinfra/anthropic/claude-4-opus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "output": [ { "amount": 82.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000825" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 200000, "maxOutputTokens": 200000, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/anthropic/claude-4-sonnet", "name": "deepinfra/anthropic/claude-4-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 200000, "maxOutputTokens": 200000, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/deepseek-ai/DeepSeek-R1", "name": "deepinfra/deepseek-ai/DeepSeek-R1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/deepseek-ai/DeepSeek-R1-0528", "name": "deepinfra/deepseek-ai/DeepSeek-R1-0528", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 2.1500000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000215" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo", "name": "deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B", "name": "deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B", "name": "deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/deepseek-ai/DeepSeek-R1-Turbo", "name": "deepinfra/deepseek-ai/DeepSeek-R1-Turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/deepseek-ai/DeepSeek-V3", "name": "deepinfra/deepseek-ai/DeepSeek-V3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "output": [ { "amount": 0.8899999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/deepseek-ai/DeepSeek-V3-0324", "name": "deepinfra/deepseek-ai/DeepSeek-V3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/deepseek-ai/DeepSeek-V3.1", "name": "deepinfra/deepseek-ai/DeepSeek-V3.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.216, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.16e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/deepseek-ai/DeepSeek-V3.1-Terminus", "name": "deepinfra/deepseek-ai/DeepSeek-V3.1-Terminus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.216, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.16e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/google/gemini-2.0-flash-001", "name": "deepinfra/google/gemini-2.0-flash-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/google/gemini-2.5-flash", "name": "deepinfra/google/gemini-2.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/google/gemini-2.5-pro", "name": "deepinfra/google/gemini-2.5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/google/gemma-3-12b-it", "name": "deepinfra/google/gemma-3-12b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/google/gemma-3-27b-it", "name": "deepinfra/google/gemma-3-27b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/google/gemma-3-4b-it", "name": "deepinfra/google/gemma-3-4b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct", "name": "deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.9e-8" } ], "output": [ { "amount": 0.049, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.9e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Llama-3.2-3B-Instruct", "name": "deepinfra/meta-llama/Llama-3.2-3B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "output": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Llama-3.3-70B-Instruct", "name": "deepinfra/meta-llama/Llama-3.3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo", "name": "deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.39, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", "name": "deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 1048576, "maxOutputTokens": 1048576, "maxTokens": 1048576, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct", "name": "deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 327680, "maxOutputTokens": 327680, "maxTokens": 327680, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Llama-Guard-3-8B", "name": "deepinfra/meta-llama/Llama-Guard-3-8B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-8" } ], "output": [ { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Llama-Guard-4-12B", "name": "deepinfra/meta-llama/Llama-Guard-4-12B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Meta-Llama-3-8B-Instruct", "name": "deepinfra/meta-llama/Meta-Llama-3-8B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "output": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct", "name": "deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo", "name": "deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct", "name": "deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo", "name": "deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/microsoft/WizardLM-2-8x22B", "name": "deepinfra/microsoft/WizardLM-2-8x22B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.8e-7" } ], "output": [ { "amount": 0.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 65536, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/microsoft/phi-4", "name": "deepinfra/microsoft/phi-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/mistralai/Mistral-Nemo-Instruct-2407", "name": "deepinfra/mistralai/Mistral-Nemo-Instruct-2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/mistralai/Mistral-Small-24B-Instruct-2501", "name": "deepinfra/mistralai/Mistral-Small-24B-Instruct-2501", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506", "name": "deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1", "name": "deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/moonshotai/Kimi-K2-Instruct", "name": "deepinfra/moonshotai/Kimi-K2-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/moonshotai/Kimi-K2-Instruct-0905", "name": "deepinfra/moonshotai/Kimi-K2-Instruct-0905", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct", "name": "deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/nvidia/Llama-3.3-Nemotron-Super-49B-v1.5", "name": "deepinfra/nvidia/Llama-3.3-Nemotron-Super-49B-v1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/nvidia/NVIDIA-Nemotron-3.5-Lightning", "name": "deepinfra/nvidia/NVIDIA-Nemotron-3.5-Lightning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 262144, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/nvidia/NVIDIA-Nemotron-Nano-9B-v2", "name": "deepinfra/nvidia/NVIDIA-Nemotron-Nano-9B-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/openai/gpt-oss-120b", "name": "deepinfra/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/openai/gpt-oss-20b", "name": "deepinfra/openai/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepinfra/zai-org/GLM-4.5", "name": "deepinfra/zai-org/GLM-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepinfra", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepseek/deepseek-chat", "name": "deepseek/deepseek-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "output": [ { "amount": 0.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.2e-7" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-8" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "deepseek/deepseek-coder", "name": "deepseek/deepseek-coder", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepseek/deepseek-r1", "name": "deepseek/deepseek-r1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "output": [ { "amount": 2.1900000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000219" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 65536, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepseek/deepseek-reasoner", "name": "deepseek/deepseek-reasoner", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "output": [ { "amount": 0.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.2e-7" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "deepseek/deepseek-v3", "name": "deepseek/deepseek-v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 65536, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepseek/deepseek-v3.2", "name": "deepseek/deepseek-v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepseek.v3-v1:0", "name": "deepseek.v3-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.58, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.8e-7" } ], "output": [ { "amount": 1.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000168" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 163840, "maxOutputTokens": 81920, "maxTokens": 81920, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepseek.v3.2", "name": "deepseek.v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.62, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.2e-7" } ], "output": [ { "amount": 1.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000185" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "dolphin", "name": "dolphin", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "nlp_cloud", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "deepseek-v3-2-251201", "name": "deepseek-v3-2-251201", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "volcengine", "maxInputTokens": 98304, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "glm-4-7-251222", "name": "glm-4-7-251222", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "volcengine", "maxInputTokens": 204800, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "kimi-k2-thinking-251104", "name": "kimi-k2-thinking-251104", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "volcengine", "maxInputTokens": 229376, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "doubao-embedding", "name": "doubao-embedding", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "volcengine", "maxInputTokens": 4096, "maxOutputTokens": null, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "doubao-embedding-large", "name": "doubao-embedding-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "volcengine", "maxInputTokens": 4096, "maxOutputTokens": null, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "doubao-embedding-large-text-240915", "name": "doubao-embedding-large-text-240915", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "volcengine", "maxInputTokens": 4096, "maxOutputTokens": null, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "doubao-embedding-large-text-250515", "name": "doubao-embedding-large-text-250515", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "volcengine", "maxInputTokens": 4096, "maxOutputTokens": null, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "doubao-embedding-text-240715", "name": "doubao-embedding-text-240715", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "volcengine", "maxInputTokens": 4096, "maxOutputTokens": null, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/search", "name": "perplexity/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "perplexity", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "searxng/search", "name": "searxng/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "searxng", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "serper/search", "name": "serper/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.001, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.001" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "serper", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "apiserpent/search", "name": "apiserpent/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0006, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.0006" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "apiserpent", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "apiserpent/deep_search", "name": "apiserpent/deep_search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0006, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.0006" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "apiserpent", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "tinyfish/search", "name": "tinyfish/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "tinyfish", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nimble/search", "name": "nimble/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "nimble", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "elevenlabs/scribe_v1", "name": "elevenlabs/scribe_v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0000611, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0000611" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "elevenlabs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "elevenlabs/scribe_v1_experimental", "name": "elevenlabs/scribe_v1_experimental", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0000611, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0000611" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "elevenlabs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "elevenlabs/eleven_v3", "name": "elevenlabs/eleven_v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00018, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00018" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "elevenlabs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "elevenlabs/eleven_multilingual_v2", "name": "elevenlabs/eleven_multilingual_v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00018, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00018" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "elevenlabs", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "embed-english-light-v2.0", "name": "embed-english-light-v2.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "cohere", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": "2026-04-04", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "embed-english-light-v3.0", "name": "embed-english-light-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "cohere", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "embed-english-v2.0", "name": "embed-english-v2.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "cohere", "maxInputTokens": 4096, "maxOutputTokens": null, "maxTokens": 4096, "deprecationDate": "2026-04-04", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "embed-english-v3.0", "name": "embed-english-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0001" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "cohere", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "embed-multilingual-v2.0", "name": "embed-multilingual-v2.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "cohere", "maxInputTokens": 768, "maxOutputTokens": null, "maxTokens": 768, "deprecationDate": "2026-04-04", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "embed-multilingual-v3.0", "name": "embed-multilingual-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "cohere", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "embed-multilingual-light-v3.0", "name": "embed-multilingual-light-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 100, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0.0001" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "cohere", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "eu.amazon.nova-lite-v1:0", "name": "eu.amazon.nova-lite-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.078, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.8e-8" } ], "output": [ { "amount": 0.312, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.12e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.amazon.nova-micro-v1:0", "name": "eu.amazon.nova-micro-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.046, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.6e-8" } ], "output": [ { "amount": 0.184, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.84e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.amazon.nova-pro-v1:0", "name": "eu.amazon.nova-pro-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.0499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000105" } ], "output": [ { "amount": 4.199999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000042" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-3-5-haiku-20241022-v1:0", "name": "eu.anthropic.claude-3-5-haiku-20241022-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [ { "amount": 0.3125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.125e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-haiku-4-5-20251001-v1:0", "name": "eu.anthropic.claude-haiku-4-5-20251001-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" }, { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": "2026-10-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-3-5-sonnet-20240620-v1:0", "name": "eu.anthropic.claude-3-5-sonnet-20240620-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-3-5-sonnet-20241022-v2:0", "name": "eu.anthropic.claude-3-5-sonnet-20241022-v2:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-3-7-sonnet-20250219-v1:0", "name": "eu.anthropic.claude-3-7-sonnet-20250219-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-3-haiku-20240307-v1:0", "name": "eu.anthropic.claude-3-haiku-20240307-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [ { "amount": 0.3125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.125e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-09-10", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-3-opus-20240229-v1:0", "name": "eu.anthropic.claude-3-opus-20240229-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-3-sonnet-20240229-v1:0", "name": "eu.anthropic.claude-3-sonnet-20240229-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-07-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-opus-4-1-20250805-v1:0", "name": "eu.anthropic.claude-opus-4-1-20250805-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-opus-4-20250514-v1:0", "name": "eu.anthropic.claude-opus-4-20250514-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-sonnet-4-20250514-v1:0", "name": "eu.anthropic.claude-sonnet-4-20250514-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": "2026-10-14", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "eu.anthropic.claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" }, { "amount": 24.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002475" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" }, { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" }, { "amount": 8.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000825" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" }, { "amount": 13.200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000132" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.meta.llama3-2-1b-instruct-v1:0", "name": "eu.meta.llama3-2-1b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "eu.meta.llama3-2-3b-instruct-v1:0", "name": "eu.meta.llama3-2-3b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.9e-7" } ], "output": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "eu.mistral.pixtral-large-2502-v1:0", "name": "eu.mistral.pixtral-large-2502-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/bria/text-to-image/3.2", "name": "fal_ai/bria/text-to-image/3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0398, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0398" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/flux-pro/v1.1", "name": "fal_ai/fal-ai/flux-pro/v1.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/flux-pro/v1.1-ultra", "name": "fal_ai/fal-ai/flux-pro/v1.1-ultra", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.06, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.06" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/flux/schnell", "name": "fal_ai/fal-ai/flux/schnell", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.003, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.003" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/bytedance/seedream/v3/text-to-image", "name": "fal_ai/fal-ai/bytedance/seedream/v3/text-to-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.03, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.03" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/bytedance/dreamina/v3.1/text-to-image", "name": "fal_ai/fal-ai/bytedance/dreamina/v3.1/text-to-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.03, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.03" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/ideogram/v3", "name": "fal_ai/fal-ai/ideogram/v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.06, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.06" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/imagen4/preview", "name": "fal_ai/fal-ai/imagen4/preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0398, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0398" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/imagen4/preview/fast", "name": "fal_ai/fal-ai/imagen4/preview/fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.02" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/imagen4/preview/ultra", "name": "fal_ai/fal-ai/imagen4/preview/ultra", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.06, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.06" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/recraft/v3/text-to-image", "name": "fal_ai/fal-ai/recraft/v3/text-to-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0398, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0398" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/stable-diffusion-v35-medium", "name": "fal_ai/fal-ai/stable-diffusion-v35-medium", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0398, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0398" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/nano-banana", "name": "fal_ai/fal-ai/nano-banana", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.039, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.039" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fal_ai/fal-ai/gemini-25-flash-image", "name": "fal_ai/fal-ai/gemini-25-flash-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.039, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.039" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fal_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks-ai-4.1b-to-16b", "name": "fireworks-ai-4.1b-to-16b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": null, "servingProvider": "fireworks_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks-ai-56b-to-176b", "name": "fireworks-ai-56b-to-176b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": null, "servingProvider": "fireworks_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks-ai-above-16b", "name": "fireworks-ai-above-16b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": null, "servingProvider": "fireworks_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks-ai-default", "name": "fireworks-ai-default", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": null, "servingProvider": "fireworks_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks-ai-embedding-150m-to-350m", "name": "fireworks-ai-embedding-150m-to-350m", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.016, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": null, "servingProvider": "fireworks_ai-embedding-models", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks-ai-embedding-up-to-150m", "name": "fireworks-ai-embedding-up-to-150m", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.008, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-9" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": null, "servingProvider": "fireworks_ai-embedding-models", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks-ai-moe-up-to-56b", "name": "fireworks-ai-moe-up-to-56b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": null, "servingProvider": "fireworks_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks-ai-up-to-4b", "name": "fireworks-ai-up-to-4b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": null, "servingProvider": "fireworks_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/WhereIsAI/UAE-Large-V1", "name": "fireworks_ai/WhereIsAI/UAE-Large-V1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.016, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.6e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "fireworks_ai-embedding-models", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-coder-v2-instruct", "name": "fireworks_ai/accounts/fireworks/models/deepseek-coder-v2-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 65536, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-r1", "name": "fireworks_ai/accounts/fireworks/models/deepseek-r1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 20480, "maxTokens": 20480, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-r1-0528", "name": "fireworks_ai/accounts/fireworks/models/deepseek-r1-0528", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 160000, "maxOutputTokens": 160000, "maxTokens": 160000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-r1-basic", "name": "fireworks_ai/accounts/fireworks/models/deepseek-r1-basic", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "output": [ { "amount": 2.1900000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000219" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 20480, "maxTokens": 20480, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-v3", "name": "fireworks_ai/accounts/fireworks/models/deepseek-v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-v3-0324", "name": "fireworks_ai/accounts/fireworks/models/deepseek-v3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-v3p1", "name": "fireworks_ai/accounts/fireworks/models/deepseek-v3p1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.6e-7" } ], "output": [ { "amount": 1.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000168" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-v3p1-terminus", "name": "fireworks_ai/accounts/fireworks/models/deepseek-v3p1-terminus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.6e-7" } ], "output": [ { "amount": 1.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000168" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-v3p2", "name": "fireworks_ai/accounts/fireworks/models/deepseek-v3p2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.6e-7" } ], "output": [ { "amount": 1.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000168" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-v4-flash", "name": "fireworks_ai/accounts/fireworks/models/deepseek-v4-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 1048576, "maxOutputTokens": 384000, "maxTokens": 384000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-v4-pro", "name": "fireworks_ai/accounts/fireworks/models/deepseek-v4-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000174" } ], "output": [ { "amount": 3.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000348" } ], "cacheRead": [ { "amount": 0.145, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.45e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 1048576, "maxOutputTokens": 384000, "maxTokens": 384000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/firefunction-v2", "name": "fireworks_ai/accounts/fireworks/models/firefunction-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/glm-4p5", "name": "fireworks_ai/accounts/fireworks/models/glm-4p5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "output": [ { "amount": 2.1900000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000219" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 96000, "maxTokens": 96000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/glm-4p5-air", "name": "fireworks_ai/accounts/fireworks/models/glm-4p5-air", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 96000, "maxTokens": 96000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/glm-4p6", "name": "fireworks_ai/accounts/fireworks/models/glm-4p6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "output": [ { "amount": 2.1900000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000219" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 202800, "maxOutputTokens": 202800, "maxTokens": 202800, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/glm-4p7", "name": "fireworks_ai/accounts/fireworks/models/glm-4p7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 202800, "maxOutputTokens": 202800, "maxTokens": 202800, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/glm-5p1", "name": "fireworks_ai/accounts/fireworks/models/glm-5p1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 202800, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/glm-5p2", "name": "fireworks_ai/accounts/fireworks/models/glm-5p2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 1048576, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/gpt-oss-120b", "name": "fireworks_ai/accounts/fireworks/models/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/gpt-oss-20b", "name": "fireworks_ai/accounts/fireworks/models/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/kimi-k2-instruct", "name": "fireworks_ai/accounts/fireworks/models/kimi-k2-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/kimi-k2-instruct-0905", "name": "fireworks_ai/accounts/fireworks/models/kimi-k2-instruct-0905", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/kimi-k2-thinking", "name": "fireworks_ai/accounts/fireworks/models/kimi-k2-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "fireworks_ai/accounts/fireworks/models/kimi-k2p5", "name": "fireworks_ai/accounts/fireworks/models/kimi-k2p5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/kimi-k2p6", "name": "fireworks_ai/accounts/fireworks/models/kimi-k2p6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.5e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/kimi-k2p7-code", "name": "fireworks_ai/accounts/fireworks/models/kimi-k2p7-code", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.5e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.9e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p1-405b-instruct", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p1-405b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p1-8b-instruct", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p1-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p2-11b-vision-instruct", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p2-11b-vision-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p2-1b-instruct", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p2-1b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p2-3b-instruct", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p2-3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p2-90b-vision-instruct", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p2-90b-vision-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama4-maverick-instruct-basic", "name": "fireworks_ai/accounts/fireworks/models/llama4-maverick-instruct-basic", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama4-scout-instruct-basic", "name": "fireworks_ai/accounts/fireworks/models/llama4-scout-instruct-basic", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/minimax-m2p1", "name": "fireworks_ai/accounts/fireworks/models/minimax-m2p1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 204800, "maxOutputTokens": 204800, "maxTokens": 204800, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/minimax-m2p7", "name": "fireworks_ai/accounts/fireworks/models/minimax-m2p7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 196608, "maxOutputTokens": 196608, "maxTokens": 196608, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/minimax-m3", "name": "fireworks_ai/accounts/fireworks/models/minimax-m3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 512000, "maxOutputTokens": 512000, "maxTokens": 512000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct-hf", "name": "fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct-hf", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 65536, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2-72b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2-72b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/yi-large", "name": "fireworks_ai/accounts/fireworks/models/yi-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/deepseek-v4-flash", "name": "fireworks_ai/deepseek-v4-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 1048576, "maxOutputTokens": 384000, "maxTokens": 384000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/deepseek-v4-pro", "name": "fireworks_ai/deepseek-v4-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000174" } ], "output": [ { "amount": 3.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000348" } ], "cacheRead": [ { "amount": 0.145, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.45e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 1048576, "maxOutputTokens": 384000, "maxTokens": 384000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/glm-4p7", "name": "fireworks_ai/glm-4p7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 202800, "maxOutputTokens": 202800, "maxTokens": 202800, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/glm-5p1", "name": "fireworks_ai/glm-5p1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 202800, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/glm-5p1-fast", "name": "fireworks_ai/glm-5p1-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000028" } ], "output": [ { "amount": 8.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000088" } ], "cacheRead": [ { "amount": 0.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.2e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 202800, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/glm-5p2", "name": "fireworks_ai/glm-5p2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 1048576, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/gpt-oss-120b", "name": "fireworks_ai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/gpt-oss-20b", "name": "fireworks_ai/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/kimi-k2p5", "name": "fireworks_ai/kimi-k2p5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/kimi-k2p6", "name": "fireworks_ai/kimi-k2p6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.5e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/kimi-k2p6-fast", "name": "fireworks_ai/kimi-k2p6-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/kimi-k2p7-code", "name": "fireworks_ai/kimi-k2p7-code", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.5e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.9e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/kimi-k2p7-code-fast", "name": "fireworks_ai/kimi-k2p7-code-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000019" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/minimax-m2p1", "name": "fireworks_ai/minimax-m2p1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 204800, "maxOutputTokens": 204800, "maxTokens": 204800, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/minimax-m2p7", "name": "fireworks_ai/minimax-m2p7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 196608, "maxOutputTokens": 196608, "maxTokens": 196608, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/minimax-m3", "name": "fireworks_ai/minimax-m3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 512000, "maxOutputTokens": 512000, "maxTokens": 512000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/qwen3p7-plus", "name": "fireworks_ai/qwen3p7-plus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/nomic-ai/nomic-embed-text-v1", "name": "fireworks_ai/nomic-ai/nomic-embed-text-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.008, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "8e-9" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "fireworks_ai-embedding-models", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/nomic-ai/nomic-embed-text-v1.5", "name": "fireworks_ai/nomic-ai/nomic-embed-text-v1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.008, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "8e-9" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "fireworks_ai-embedding-models", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/thenlper/gte-base", "name": "fireworks_ai/thenlper/gte-base", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.008, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "8e-9" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "fireworks_ai-embedding-models", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/thenlper/gte-large", "name": "fireworks_ai/thenlper/gte-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.016, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.6e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "fireworks_ai-embedding-models", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "friendliai/meta-llama-3.1-70b-instruct", "name": "friendliai/meta-llama-3.1-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "friendliai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "friendliai/meta-llama-3.1-8b-instruct", "name": "friendliai/meta-llama-3.1-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "friendliai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "ft:babbage-002", "name": "ft:babbage-002", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "text-completion-openai", "maxInputTokens": 16384, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ft:davinci-002", "name": "ft:davinci-002", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "text-completion-openai", "maxInputTokens": 16384, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ft:gpt-3.5-turbo", "name": "ft:gpt-3.5-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "ft:gpt-3.5-turbo-0125", "name": "ft:gpt-3.5-turbo-0125", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "ft:gpt-3.5-turbo-0613", "name": "ft:gpt-3.5-turbo-0613", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "ft:gpt-3.5-turbo-1106", "name": "ft:gpt-3.5-turbo-1106", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "ft:gpt-4-0613", "name": "ft:gpt-4-0613", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "ft:gpt-4o-2024-08-06", "name": "ft:gpt-4o-2024-08-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "ft:gpt-4o-2024-11-20", "name": "ft:gpt-4o-2024-11-20", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "ft:gpt-4o-mini-2024-07-18", "name": "ft:gpt-4o-mini-2024-07-18", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "ft:gpt-4.1-2025-04-14", "name": "ft:gpt-4.1-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "ft:gpt-4.1-mini-2025-04-14", "name": "ft:gpt-4.1-mini-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" }, { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "ft:gpt-4.1-nano-2025-04-14", "name": "ft:gpt-4.1-nano-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "ft:o4-mini-2025-04-16", "name": "ft:o4-mini-2025-04-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000016" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "gemini-2.0-flash", "name": "gemini-2.0-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": true, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.0-flash-001", "name": "gemini-2.0-flash-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": true, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.0-flash-lite", "name": "gemini-2.0-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.01875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.875e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": null, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": true, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.0-flash-lite-001", "name": "gemini-2.0-flash-lite-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.01875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.875e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": null, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": true, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.5-flash", "name": "gemini-2.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.5-flash-image", "name": "gemini-2.5-flash-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "3e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" }, { "amount": 0.039, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.039" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": false } } }, { "id": "gemini-3-pro-image", "name": "gemini-3-pro-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000006" }, { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0011, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0011" }, { "amount": 0.134, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.134" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3-pro-image-preview", "name": "gemini-3-pro-image-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000006" }, { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0011, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0011" }, { "amount": 0.134, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.134" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3.1-flash-image", "name": "gemini-3.1-flash-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "5e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00056, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00056" }, { "amount": 0.0672, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0672" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3.1-flash-image-preview", "name": "gemini-3.1-flash-image-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "5e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00056, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00056" }, { "amount": 0.0672, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0672" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3.1-flash-lite-preview", "name": "gemini-3.1-flash-lite-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3.1-flash-lite", "name": "gemini-3.1-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" }, { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3.5-flash-lite", "name": "gemini-3.5-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.4e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" }, { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "deep-research-pro-preview-12-2025", "name": "deep-research-pro-preview-12-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000006" }, { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0011, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0011" }, { "amount": 0.134, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.134" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.5-flash-lite", "name": "gemini-2.5-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-8" } ], "cacheWrite": [], "other": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.5-flash-lite-preview-09-2025", "name": "gemini-2.5-flash-lite-preview-09-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-8" } ], "cacheWrite": [], "other": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.5-flash-preview-09-2025", "name": "gemini-2.5-flash-preview-09-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-live-2.5-flash-preview-native-audio-09-2025", "name": "gemini-live-2.5-flash-preview-native-audio-09-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" }, { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000003" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000002" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": true, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-live-2.5-flash-preview-native-audio-09-2025", "name": "gemini/gemini-live-2.5-flash-preview-native-audio-09-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" }, { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000003" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000002" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": true, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.5-flash-lite-preview-06-17", "name": "gemini-2.5-flash-lite-preview-06-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "5e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": "2025-11-18", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.5-pro", "name": "gemini-2.5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3-pro-preview", "name": "gemini-3-pro-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": "2026-03-26", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3.1-pro-preview", "name": "gemini-3.1-pro-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "other": [ { "amount": 0.00012, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00012" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3.1-pro-preview-customtools", "name": "gemini-3.1-pro-preview-customtools", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "other": [ { "amount": 0.00012, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00012" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "vertex_ai/gemini-3-pro-preview", "name": "vertex_ai/gemini-3-pro-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "vertex_ai/gemini-3-flash-preview", "name": "vertex_ai/gemini-3-flash-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" }, { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "vertex_ai/gemini-3.5-flash", "name": "vertex_ai/gemini-3.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" }, { "amount": 16.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000162" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "cacheWrite": [], "other": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "vertex_ai/gemini-3.6-flash", "name": "vertex_ai/gemini-3.6-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 13.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000135" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "cacheWrite": [], "other": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "vertex_ai/gemini-3.7-flash", "name": "vertex_ai/gemini-3.7-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" }, { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" }, { "amount": 6.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000675" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.35e-7" } ], "cacheWrite": [], "other": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "vertex_ai/gemini-3.1-pro-preview", "name": "vertex_ai/gemini-3.1-pro-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "other": [ { "amount": 0.00012, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00012" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "vertex_ai/gemini-3.1-pro-preview-customtools", "name": "vertex_ai/gemini-3.1-pro-preview-customtools", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "other": [ { "amount": 0.00012, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00012" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.5-pro-preview-tts", "name": "gemini-2.5-pro-preview-tts", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7e-7" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-robotics-er-1.5-preview", "name": "gemini-robotics-er-1.5-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": false, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gemini/gemini-robotics-er-1.5-preview", "name": "gemini/gemini-robotics-er-1.5-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": "2026-04-30", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": false, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-robotics-er-2-preview", "name": "gemini/gemini-robotics-er-2-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [], "other": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-robotics-er-1.6-preview", "name": "gemini/gemini-robotics-er-1.6-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.5-computer-use-preview-10-2025", "name": "gemini-2.5-computer-use-preview-10-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 128000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gemini-embedding-001", "name": "gemini-embedding-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.5e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai-embedding-models", "maxInputTokens": 2048, "maxOutputTokens": null, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini-embedding-2-preview", "name": "gemini-embedding-2-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00016, "currency": "USD", "units": 1, "pricingType": "second", "modality": "audio", "raw": "0.00016" }, { "amount": 0.00079, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.00079" }, { "amount": 0.00012, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00012" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai-embedding-models", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini-embedding-2", "name": "gemini-embedding-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00016, "currency": "USD", "units": 1, "pricingType": "second", "modality": "audio", "raw": "0.00016" }, { "amount": 0.00079, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.00079" }, { "amount": 0.00012, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00012" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai-embedding-models", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/gemini-embedding-2-preview", "name": "vertex_ai/gemini-embedding-2-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00016, "currency": "USD", "units": 1, "pricingType": "second", "modality": "audio", "raw": "0.00016" }, { "amount": 0.00079, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.00079" }, { "amount": 0.00012, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00012" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/gemini-embedding-2", "name": "vertex_ai/gemini-embedding-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00016, "currency": "USD", "units": 1, "pricingType": "second", "modality": "audio", "raw": "0.00016" }, { "amount": 0.00079, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.00079" }, { "amount": 0.00012, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00012" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini-flash-experimental", "name": "gemini-flash-experimental", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-embedding-001", "name": "gemini/gemini-embedding-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.5e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "gemini", "maxInputTokens": 2048, "maxOutputTokens": null, "maxTokens": 2048, "deprecationDate": "2028-05-14", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-embedding-2-preview", "name": "gemini/gemini-embedding-2-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00016, "currency": "USD", "units": 1, "pricingType": "second", "modality": "audio", "raw": "0.00016" }, { "amount": 0.00079, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.00079" }, { "amount": 0.00012, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00012" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "gemini", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": "2026-08-10", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-embedding-2", "name": "gemini/gemini-embedding-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00016, "currency": "USD", "units": 1, "pricingType": "second", "modality": "audio", "raw": "0.00016" }, { "amount": 0.00079, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.00079" }, { "amount": 0.00012, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00012" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "gemini", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-1.5-flash", "name": "gemini/gemini-1.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "7.5e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "gemini", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": "2025-09-29", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-2.0-flash", "name": "gemini/gemini-2.0-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": true, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-2.0-flash-001", "name": "gemini/gemini-2.0-flash-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-2.0-flash-lite", "name": "gemini/gemini-2.0-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.01875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.875e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": null, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": true, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-2.5-flash", "name": "gemini/gemini-2.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-2.5-flash-image", "name": "gemini/gemini-2.5-flash-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "3e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" }, { "amount": 0.039, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.039" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-10-02", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": false, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3-pro-image", "name": "gemini/gemini-3-pro-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000006" }, { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0011, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0011" }, { "amount": 0.134, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.134" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": false, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3-pro-image-preview", "name": "gemini/gemini-3-pro-image-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000006" }, { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0011, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0011" }, { "amount": 0.134, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.134" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-06-25", "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": false, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3.1-flash-image", "name": "gemini/gemini-3.1-flash-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "2.5e-7" }, { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.25e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000015" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "7.5e-7" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.045, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.045" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": false, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3.1-flash-image-preview", "name": "gemini/gemini-3.1-flash-image-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "2.5e-7" }, { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.25e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000015" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "7.5e-7" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.045, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.045" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-06-25", "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": false, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/deep-research-pro-preview-12-2025", "name": "gemini/deep-research-pro-preview-12-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000006" }, { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0011, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0011" }, { "amount": 0.134, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.134" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-2.5-flash-lite", "name": "gemini/gemini-2.5-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-8" } ], "cacheWrite": [], "other": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-2.5-flash-lite-preview-09-2025", "name": "gemini/gemini-2.5-flash-lite-preview-09-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-8" } ], "cacheWrite": [], "other": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": "2026-03-31", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-2.5-flash-preview-09-2025", "name": "gemini/gemini-2.5-flash-preview-09-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": "2026-02-17", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-flash-latest", "name": "gemini/gemini-flash-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-flash-lite-latest", "name": "gemini/gemini-flash-lite-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-2.5-flash-lite-preview-06-17", "name": "gemini/gemini-2.5-flash-lite-preview-06-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "5e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": "2025-11-18", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-2.5-flash-preview-tts", "name": "gemini/gemini-2.5-flash-preview-tts", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "gemini", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-2.5-pro", "name": "gemini/gemini-2.5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-2.5-computer-use-preview-10-2025", "name": "gemini/gemini-2.5-computer-use-preview-10-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 128000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gemini/gemini-3-pro-preview", "name": "gemini/gemini-3-pro-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": "2026-03-09", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3.1-flash-lite-preview", "name": "gemini/gemini-3.1-flash-lite-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": "2026-05-25", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3.1-flash-lite", "name": "gemini/gemini-3.1-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" }, { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": "2027-05-07", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3.5-flash-lite", "name": "gemini/gemini-3.5-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.4e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" }, { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3-flash-preview", "name": "gemini/gemini-3-flash-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" }, { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "cacheWrite": [], "other": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3.5-flash", "name": "gemini/gemini-3.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" }, { "amount": 16.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000162" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "cacheWrite": [], "other": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3.6-flash", "name": "gemini/gemini-3.6-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 13.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000135" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "cacheWrite": [], "other": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3.7-flash", "name": "gemini/gemini-3.7-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" }, { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" }, { "amount": 6.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000675" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.35e-7" } ], "cacheWrite": [], "other": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-omni-flash-preview", "name": "gemini/gemini-omni-flash-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000015" } ], "output": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gemini/gemini-3.1-pro-preview", "name": "gemini/gemini-3.1-pro-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-3.1-pro-preview-customtools", "name": "gemini/gemini-3.1-pro-preview-customtools", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 21.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000216" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3-flash-preview", "name": "gemini-3-flash-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" }, { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "cacheWrite": [], "other": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-omni-flash-preview", "name": "gemini-omni-flash-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000015" } ], "output": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gemini-3.5-flash", "name": "gemini-3.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" }, { "amount": 16.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000162" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "cacheWrite": [], "other": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3.6-flash", "name": "gemini-3.6-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 13.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000135" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "cacheWrite": [], "other": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-3.7-flash", "name": "gemini-3.7-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" }, { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" }, { "amount": 6.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000675" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.35e-7" } ], "cacheWrite": [], "other": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-2.5-pro-preview-tts", "name": "gemini/gemini-2.5-pro-preview-tts", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7e-7" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-exp-1114", "name": "gemini/gemini-exp-1114", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gemini/gemini-exp-1206", "name": "gemini/gemini-exp-1206", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 2097152, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gemini/gemini-gemma-2-27b-it", "name": "gemini/gemini-gemma-2-27b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 1.0499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000105" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": null, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-gemma-2-9b-it", "name": "gemini/gemini-gemma-2-9b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 1.0499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000105" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": null, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemma-3-27b-it", "name": "gemini/gemma-3-27b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "modality": "audio", "raw": "0" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": false, "webSearch": null } } }, { "id": "gemini/imagen-3.0-fast-generate-001", "name": "gemini/imagen-3.0-fast-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.02" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/imagen-3.0-generate-001", "name": "gemini/imagen-3.0-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/imagen-3.0-generate-002", "name": "gemini/imagen-3.0-generate-002", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2025-11-10", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/imagen-4.0-fast-generate-001", "name": "gemini/imagen-4.0-fast-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.02" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-08-17", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/imagen-4.0-generate-001", "name": "gemini/imagen-4.0-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-08-17", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/imagen-4.0-ultra-generate-001", "name": "gemini/imagen-4.0-ultra-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.06, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.06" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-08-17", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/learnlm-1.5-pro-experimental", "name": "gemini/learnlm-1.5-pro-experimental", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "modality": "audio", "raw": "0" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 32767, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gemini/lyria-3-clip-preview", "name": "gemini/lyria-3-clip-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": false, "audioInput": false, "audioOutput": true, "promptCaching": false, "reasoning": null, "responseSchema": false, "systemMessages": false, "webSearch": false } } }, { "id": "gemini/lyria-3-pro-preview", "name": "gemini/lyria-3-pro-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": false, "audioInput": false, "audioOutput": true, "promptCaching": false, "reasoning": null, "responseSchema": false, "systemMessages": false, "webSearch": false } } }, { "id": "gemini/veo-2.0-generate-001", "name": "gemini/veo-2.0-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.35, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.35" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "gemini", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": "2026-06-30", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/veo-3.1-fast-generate-preview", "name": "gemini/veo-3.1-fast-generate-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.15, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.15" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "gemini", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/veo-3.1-generate-preview", "name": "gemini/veo-3.1-generate-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.4, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.4" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "gemini", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/veo-3.1-lite-generate-preview", "name": "gemini/veo-3.1-lite-generate-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.05, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.05" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "gemini", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/veo-3.1-fast-generate-001", "name": "gemini/veo-3.1-fast-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.15, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.15" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "gemini", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/veo-3.1-generate-001", "name": "gemini/veo-3.1-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.4, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.4" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "gemini", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "github_copilot/mai-code-1-flash", "name": "github_copilot/mai-code-1-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "github_copilot", "maxInputTokens": 128000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "github_copilot/mai-code-1-flash-internal", "name": "github_copilot/mai-code-1-flash-internal", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "github_copilot", "maxInputTokens": 128000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "gigachat/GigaChat-2-Lite", "name": "gigachat/GigaChat-2-Lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gigachat", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gigachat/GigaChat-2-Max", "name": "gigachat/GigaChat-2-Max", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gigachat", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gigachat/GigaChat-2-Pro", "name": "gigachat/GigaChat-2-Pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gigachat", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gigachat/Embeddings", "name": "gigachat/Embeddings", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "gigachat", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gigachat/Embeddings-2", "name": "gigachat/Embeddings-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "gigachat", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gigachat/EmbeddingsGigaR", "name": "gigachat/EmbeddingsGigaR", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "gigachat", "maxInputTokens": 4096, "maxOutputTokens": null, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/anthropic/claude-opus-4.5", "name": "gmi/anthropic/claude-opus-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 409600, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/anthropic/claude-sonnet-4.5", "name": "gmi/anthropic/claude-sonnet-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 409600, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/anthropic/claude-sonnet-4", "name": "gmi/anthropic/claude-sonnet-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 409600, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/anthropic/claude-opus-4", "name": "gmi/anthropic/claude-opus-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 409600, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/openai/gpt-5.2", "name": "gmi/openai/gpt-5.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 409600, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/openai/gpt-5.1", "name": "gmi/openai/gpt-5.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 409600, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/openai/gpt-5", "name": "gmi/openai/gpt-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 409600, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/openai/gpt-4o", "name": "gmi/openai/gpt-4o", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 131072, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/openai/gpt-4o-mini", "name": "gmi/openai/gpt-4o-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 131072, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/deepseek-ai/DeepSeek-V3.2", "name": "gmi/deepseek-ai/DeepSeek-V3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 163840, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/deepseek-ai/DeepSeek-V3-0324", "name": "gmi/deepseek-ai/DeepSeek-V3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 163840, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/google/gemini-3-pro-preview", "name": "gmi/google/gemini-3-pro-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/google/gemini-3-flash-preview", "name": "gmi/google/gemini-3-flash-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/moonshotai/Kimi-K2-Thinking", "name": "gmi/moonshotai/Kimi-K2-Thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 262144, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/MiniMaxAI/MiniMax-M2.1", "name": "gmi/MiniMaxAI/MiniMax-M2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 196608, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "baseten/MiniMaxAI/MiniMax-M2.5", "name": "baseten/MiniMaxAI/MiniMax-M2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "baseten", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "baseten/nvidia/Nemotron-120B-A12B", "name": "baseten/nvidia/Nemotron-120B-A12B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "baseten", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "baseten/zai-org/GLM-5", "name": "baseten/zai-org/GLM-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.5e-7" } ], "output": [ { "amount": 3.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000315" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "baseten", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "baseten/zai-org/GLM-4.7", "name": "baseten/zai-org/GLM-4.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "baseten", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "baseten/zai-org/GLM-4.6", "name": "baseten/zai-org/GLM-4.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "baseten", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "baseten/moonshotai/Kimi-K2.5", "name": "baseten/moonshotai/Kimi-K2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "baseten", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "baseten/moonshotai/Kimi-K2-Thinking", "name": "baseten/moonshotai/Kimi-K2-Thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "baseten", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "baseten/moonshotai/Kimi-K2-Instruct-0905", "name": "baseten/moonshotai/Kimi-K2-Instruct-0905", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "baseten", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "baseten/openai/gpt-oss-120b", "name": "baseten/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "baseten", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "baseten/deepseek-ai/DeepSeek-V3.1", "name": "baseten/deepseek-ai/DeepSeek-V3.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "baseten", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "baseten/deepseek-ai/DeepSeek-V3-0324", "name": "baseten/deepseek-ai/DeepSeek-V3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.77, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.7e-7" } ], "output": [ { "amount": 0.77, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "baseten", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/Qwen/Qwen3-VL-235B-A22B-Instruct-FP8", "name": "gmi/Qwen/Qwen3-VL-235B-A22B-Instruct-FP8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 262144, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gmi/zai-org/GLM-4.7-FP8", "name": "gmi/zai-org/GLM-4.7-FP8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gmi", "maxInputTokens": 202752, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "google.gemma-3-12b-it", "name": "google.gemma-3-12b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "google.gemma-3-27b-it", "name": "google.gemma-3-27b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.3e-7" } ], "output": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "google.gemma-3-4b-it", "name": "google.gemma-3-4b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "google_pse/search", "name": "google_pse/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "google_pse", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "global.anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "global.anthropic.claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "global.anthropic.claude-sonnet-4-20250514-v1:0", "name": "global.anthropic.claude-sonnet-4-20250514-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": "2026-10-14", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "global.anthropic.claude-haiku-4-5-20251001-v1:0", "name": "global.anthropic.claude-haiku-4-5-20251001-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "global.amazon.nova-2-lite-v1:0", "name": "global.amazon.nova-2-lite-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-3.5-turbo", "name": "gpt-3.5-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-3.5-turbo-0125", "name": "gpt-3.5-turbo-0125", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-3.5-turbo-1106", "name": "gpt-3.5-turbo-1106", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-09-28", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-3.5-turbo-16k", "name": "gpt-3.5-turbo-16k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-3.5-turbo-instruct", "name": "gpt-3.5-turbo-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "text-completion-openai", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-3.5-turbo-instruct-0914", "name": "gpt-3.5-turbo-instruct-0914", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "text-completion-openai", "maxInputTokens": 8192, "maxOutputTokens": 4097, "maxTokens": 4097, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-4", "name": "gpt-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4-0125-preview", "name": "gpt-4-0125-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-03-26", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4-0314", "name": "gpt-4-0314", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-03-26", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4-0613", "name": "gpt-4-0613", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4-1106-preview", "name": "gpt-4-1106-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4-turbo", "name": "gpt-4-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4-turbo-2024-04-09", "name": "gpt-4-turbo-2024-04-09", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4-turbo-preview", "name": "gpt-4-turbo-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-03-26", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4.1", "name": "gpt-4.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" }, { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-4.1-2025-04-14", "name": "gpt-4.1-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" }, { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-4.1-mini", "name": "gpt-4.1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" }, { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" }, { "amount": 2.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000028" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-4.1-mini-2025-04-14", "name": "gpt-4.1-mini-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" }, { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" }, { "amount": 2.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000028" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-4.1-nano", "name": "gpt-4.1-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" }, { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4.1-nano-2025-04-14", "name": "gpt-4.1-nano-2025-04-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" }, { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o", "name": "gpt-4o", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 4.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000425" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 17, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000017" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-2024-05-13", "name": "gpt-4o-2024-05-13", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 8.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000875" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 26.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002625" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-2024-08-06", "name": "gpt-4o-2024-08-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 4.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000425" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 17, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000017" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-2024-11-20", "name": "gpt-4o-2024-11-20", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 4.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000425" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 17, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000017" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-audio-preview", "name": "gpt-4o-audio-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00004" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-audio-preview-2024-12-17", "name": "gpt-4o-audio-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00004" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-audio-preview-2025-06-03", "name": "gpt-4o-audio-preview-2025-06-03", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00004" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-audio", "name": "gpt-audio", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000032" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000064" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": true, "audioOutput": true, "promptCaching": false, "reasoning": false, "responseSchema": false, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-audio-1.5", "name": "gpt-audio-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000032" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000064" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": true, "audioOutput": true, "promptCaching": false, "reasoning": false, "responseSchema": false, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-audio-2025-08-28", "name": "gpt-audio-2025-08-28", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000032" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000064" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": true, "audioOutput": true, "promptCaching": false, "reasoning": false, "responseSchema": false, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-audio-mini", "name": "gpt-audio-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": true, "audioOutput": true, "promptCaching": false, "reasoning": false, "responseSchema": false, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-audio-mini-2025-10-06", "name": "gpt-audio-mini-2025-10-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": true, "audioOutput": true, "promptCaching": false, "reasoning": false, "responseSchema": false, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-audio-mini-2025-12-15", "name": "gpt-audio-mini-2025-12-15", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": true, "audioOutput": true, "promptCaching": false, "reasoning": false, "responseSchema": false, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-mini", "name": "gpt-4o-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-mini-2024-07-18", "name": "gpt-4o-mini-2024-07-18", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-mini-audio-preview", "name": "gpt-4o-mini-audio-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-mini-audio-preview-2024-12-17", "name": "gpt-4o-mini-audio-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-mini-realtime-preview", "name": "gpt-4o-mini-realtime-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-mini-realtime-preview-2024-12-17", "name": "gpt-4o-mini-realtime-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-mini-search-preview", "name": "gpt-4o-mini-search-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-4o-mini-search-preview-2025-03-11", "name": "gpt-4o-mini-search-preview-2025-03-11", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-mini-transcribe", "name": "gpt-4o-mini-transcribe", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000125" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000125" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "openai", "maxInputTokens": 16000, "maxOutputTokens": 2000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-4o-mini-tts", "name": "gpt-4o-mini-tts", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00025, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00025" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-4o-realtime-preview", "name": "gpt-4o-realtime-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000005" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00004" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-realtime-preview-2024-12-17", "name": "gpt-4o-realtime-preview-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000005" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00004" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-realtime-preview-2025-06-03", "name": "gpt-4o-realtime-preview-2025-06-03", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000005" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00004" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" }, { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00008" } ], "cacheRead": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-search-preview", "name": "gpt-4o-search-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-4o-search-preview-2025-03-11", "name": "gpt-4o-search-preview-2025-03-11", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-4o-transcribe", "name": "gpt-4o-transcribe", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "openai", "maxInputTokens": 16000, "maxOutputTokens": 2000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-image-1.5", "name": "gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000032" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-12-01", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-image-1.5-2025-12-16", "name": "gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000032" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-12-01", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-image-2", "name": "gpt-image-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00003" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-image-2-2026-04-21", "name": "gpt-image-2-2026-04-21", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00003" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "low/1024-x-1024/gpt-image-1.5", "name": "low/1024-x-1024/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.009, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.009" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "low/1024-x-1536/gpt-image-1.5", "name": "low/1024-x-1536/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "low/1536-x-1024/gpt-image-1.5", "name": "low/1536-x-1024/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1024-x-1024/gpt-image-1.5", "name": "medium/1024-x-1024/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.034, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.034" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1024-x-1536/gpt-image-1.5", "name": "medium/1024-x-1536/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.05, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.05" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1536-x-1024/gpt-image-1.5", "name": "medium/1536-x-1024/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.05, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.05" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "high/1024-x-1024/gpt-image-1.5", "name": "high/1024-x-1024/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.133, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.133" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "high/1024-x-1536/gpt-image-1.5", "name": "high/1024-x-1536/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.2, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.2" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "high/1536-x-1024/gpt-image-1.5", "name": "high/1536-x-1024/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.2, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.2" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "standard/1024-x-1024/gpt-image-1.5", "name": "standard/1024-x-1024/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.009, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.009" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "standard/1024-x-1536/gpt-image-1.5", "name": "standard/1024-x-1536/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "standard/1536-x-1024/gpt-image-1.5", "name": "standard/1536-x-1024/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "1024-x-1024/gpt-image-1.5", "name": "1024-x-1024/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.009, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.009" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "1024-x-1536/gpt-image-1.5", "name": "1024-x-1536/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "1536-x-1024/gpt-image-1.5", "name": "1536-x-1024/gpt-image-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "low/1024-x-1024/gpt-image-1.5-2025-12-16", "name": "low/1024-x-1024/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.009, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.009" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "low/1024-x-1536/gpt-image-1.5-2025-12-16", "name": "low/1024-x-1536/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "low/1536-x-1024/gpt-image-1.5-2025-12-16", "name": "low/1536-x-1024/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1024-x-1024/gpt-image-1.5-2025-12-16", "name": "medium/1024-x-1024/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.034, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.034" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1024-x-1536/gpt-image-1.5-2025-12-16", "name": "medium/1024-x-1536/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.05, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.05" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1536-x-1024/gpt-image-1.5-2025-12-16", "name": "medium/1536-x-1024/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.05, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.05" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "high/1024-x-1024/gpt-image-1.5-2025-12-16", "name": "high/1024-x-1024/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.133, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.133" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "high/1024-x-1536/gpt-image-1.5-2025-12-16", "name": "high/1024-x-1536/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.2, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.2" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "high/1536-x-1024/gpt-image-1.5-2025-12-16", "name": "high/1536-x-1024/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.2, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.2" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "standard/1024-x-1024/gpt-image-1.5-2025-12-16", "name": "standard/1024-x-1024/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.009, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.009" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "standard/1024-x-1536/gpt-image-1.5-2025-12-16", "name": "standard/1024-x-1536/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "standard/1536-x-1024/gpt-image-1.5-2025-12-16", "name": "standard/1536-x-1024/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "1024-x-1024/gpt-image-1.5-2025-12-16", "name": "1024-x-1024/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.009, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.009" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "1024-x-1536/gpt-image-1.5-2025-12-16", "name": "1024-x-1536/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "1536-x-1024/gpt-image-1.5-2025-12-16", "name": "1536-x-1024/gpt-image-1.5-2025-12-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.013, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.013" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-5", "name": "gpt-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.1", "name": "gpt-5.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.1-2025-11-13", "name": "gpt-5.1-2025-11-13", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.1-chat-latest", "name": "gpt-5.1-chat-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-07-23", "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.2", "name": "gpt-5.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" }, { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.2-2025-12-11", "name": "gpt-5.2-2025-12-11", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" }, { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.2-chat-latest", "name": "gpt-5.2-chat-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" }, { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-08-10", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.3-chat-latest", "name": "gpt-5.3-chat-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" }, { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-08-10", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.2-pro", "name": "gpt-5.2-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000021" } ], "output": [ { "amount": 168, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000168" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.2-pro-2025-12-11", "name": "gpt-5.2-pro-2025-12-11", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000021" } ], "output": [ { "amount": 168, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000168" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.6", "name": "gpt-5.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.6-sol", "name": "gpt-5.6-sol", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.6-terra", "name": "gpt-5.6-terra", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.6-luna", "name": "gpt-5.6-luna", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" }, { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" }, { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" }, { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" }, { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.5", "name": "gpt-5.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.5-2026-04-23", "name": "gpt-5.5-2026-04-23", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" }, { "amount": 45, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000045" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.5-pro", "name": "gpt-5.5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" }, { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00027" } ], "cacheRead": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.5-pro-2026-04-23", "name": "gpt-5.5-pro-2026-04-23", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" }, { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00027" } ], "cacheRead": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.4", "name": "gpt-5.4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-5.4-2026-03-05", "name": "gpt-5.4-2026-03-05", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-5.4-pro", "name": "gpt-5.4-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" }, { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00027" } ], "cacheRead": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.4-pro-2026-03-05", "name": "gpt-5.4-pro-2026-03-05", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "output": [ { "amount": 180, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00018" }, { "amount": 90, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00009" }, { "amount": 270, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00027" } ], "cacheRead": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 1050000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.4-mini", "name": "gpt-5.4-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" }, { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" }, { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.4-mini-2026-03-17", "name": "gpt-5.4-mini-2026-03-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" }, { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" }, { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.4-nano", "name": "gpt-5.4-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.25e-7" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5.4-nano-2026-03-17", "name": "gpt-5.4-nano-2026-03-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 0.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.25e-7" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5-pro", "name": "gpt-5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "output": [ { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00012" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 400000, "maxOutputTokens": 272000, "maxTokens": 272000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5-pro-2025-10-06", "name": "gpt-5-pro-2025-10-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "output": [ { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00012" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 400000, "maxOutputTokens": 272000, "maxTokens": 272000, "deprecationDate": "2026-12-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5-2025-08-07", "name": "gpt-5-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-12-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5-chat", "name": "gpt-5-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-5-chat-latest", "name": "gpt-5-chat-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-07-23", "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-5-codex", "name": "gpt-5-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": true } } }, { "id": "gpt-5.1-codex", "name": "gpt-5.1-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": true } } }, { "id": "gpt-5.1-codex-max", "name": "gpt-5.1-codex-max", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": true } } }, { "id": "gpt-5.1-codex-mini", "name": "gpt-5.1-codex-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" }, { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": true } } }, { "id": "gpt-5.2-codex", "name": "gpt-5.2-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" }, { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": true } } }, { "id": "gpt-5.3-codex", "name": "gpt-5.3-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" }, { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" }, { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": false, "webSearch": true } } }, { "id": "gpt-5-mini", "name": "gpt-5-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" }, { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5-mini-2025-08-07", "name": "gpt-5-mini-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" }, { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-12-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5-nano", "name": "gpt-5-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-9" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5-nano-2025-08-07", "name": "gpt-5-nano-2025-08-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-9" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-12-11", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-image-1", "name": "gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" } ], "output": [ { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00004" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-10-23", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-image-1-mini", "name": "gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "2e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-12-01", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-realtime", "name": "gpt-realtime", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000032" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000016" }, { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000064" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "4e-7" } ], "cacheWrite": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "4e-7" } ], "other": [ { "amount": 0.000005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000005" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 32000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-realtime-1.5", "name": "gpt-realtime-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000032" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000016" }, { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000064" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "4e-7" } ], "cacheWrite": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "4e-7" } ], "other": [ { "amount": 0.000005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000005" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 32000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-realtime-2", "name": "gpt-realtime-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000032" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000016" }, { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000064" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "4e-7" } ], "cacheWrite": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "4e-7" } ], "other": [ { "amount": 0.000005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000005" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 32000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-realtime-2.1", "name": "gpt-realtime-2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000032" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000024" }, { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000064" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "4e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "4e-7" } ], "cacheWrite": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "4e-7" } ], "other": [ { "amount": 0.000005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000005" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-realtime-2.1-mini", "name": "gpt-realtime-2.1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-8" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "other": [ { "amount": 8e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "8e-7" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-realtime-mini", "name": "gpt-realtime-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-realtime-2025-08-28", "name": "gpt-realtime-2025-08-28", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000004" }, { "amount": 32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000032" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000016" }, { "amount": 64, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000064" } ], "cacheRead": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "4e-7" } ], "cacheWrite": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "4e-7" } ], "other": [ { "amount": 0.000005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000005" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 32000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2027-01-20", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gradient_ai/anthropic-claude-3-opus", "name": "gradient_ai/anthropic-claude-3-opus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gradient_ai", "maxInputTokens": 200000, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gradient_ai/anthropic-claude-3.5-haiku", "name": "gradient_ai/anthropic-claude-3.5-haiku", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gradient_ai", "maxInputTokens": 200000, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gradient_ai/anthropic-claude-3.5-sonnet", "name": "gradient_ai/anthropic-claude-3.5-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gradient_ai", "maxInputTokens": 200000, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gradient_ai/anthropic-claude-3.7-sonnet", "name": "gradient_ai/anthropic-claude-3.7-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gradient_ai", "maxInputTokens": 200000, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gradient_ai/deepseek-r1-distill-llama-70b", "name": "gradient_ai/deepseek-r1-distill-llama-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.9900000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.9e-7" } ], "output": [ { "amount": 0.9900000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gradient_ai", "maxInputTokens": 32768, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gradient_ai/llama3-8b-instruct", "name": "gradient_ai/llama3-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gradient_ai", "maxInputTokens": 8192, "maxOutputTokens": 512, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gradient_ai/llama3.3-70b-instruct", "name": "gradient_ai/llama3.3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "output": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gradient_ai", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gradient_ai/mistral-nemo-instruct-2407", "name": "gradient_ai/mistral-nemo-instruct-2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gradient_ai", "maxInputTokens": 128000, "maxOutputTokens": 512, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gradient_ai/openai-o3", "name": "gradient_ai/openai-o3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gradient_ai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gradient_ai/openai-o3-mini", "name": "gradient_ai/openai-o3-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gradient_ai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "lemonade/Qwen3-Coder-30B-A3B-Instruct-GGUF", "name": "lemonade/Qwen3-Coder-30B-A3B-Instruct-GGUF", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lemonade", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "lemonade/gpt-oss-20b-mxfp4-GGUF", "name": "lemonade/gpt-oss-20b-mxfp4-GGUF", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lemonade", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "lemonade/gpt-oss-120b-mxfp-GGUF", "name": "lemonade/gpt-oss-120b-mxfp-GGUF", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lemonade", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "lemonade/Gemma-3-4b-it-GGUF", "name": "lemonade/Gemma-3-4b-it-GGUF", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lemonade", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "lemonade/Qwen3-4B-Instruct-2507-GGUF", "name": "lemonade/Qwen3-4B-Instruct-2507-GGUF", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lemonade", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "amazon-nova/nova-micro-v1", "name": "amazon-nova/nova-micro-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-8" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "amazon_nova", "maxInputTokens": 128000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "amazon-nova/nova-lite-v1", "name": "amazon-nova/nova-lite-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "amazon_nova", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "amazon-nova/nova-premier-v1", "name": "amazon-nova/nova-premier-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "amazon_nova", "maxInputTokens": 1000000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "amazon-nova/nova-pro-v1", "name": "amazon-nova/nova-pro-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "amazon_nova", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "groq/llama-3.1-8b-instant", "name": "groq/llama-3.1-8b-instant", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2026-08-16", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "groq/llama-3.3-70b-versatile", "name": "groq/llama-3.3-70b-versatile", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.59, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.9e-7" } ], "output": [ { "amount": 0.7899999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-08-16", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "groq/gemma-7b-it", "name": "groq/gemma-7b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "groq/meta-llama/llama-prompt-guard-2-22m", "name": "groq/meta-llama/llama-prompt-guard-2-22m", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 512, "maxOutputTokens": 512, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "groq/meta-llama/llama-prompt-guard-2-86m", "name": "groq/meta-llama/llama-prompt-guard-2-86m", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 512, "maxOutputTokens": 512, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "groq/meta-llama/llama-guard-4-12b", "name": "groq/meta-llama/llama-guard-4-12b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-03-05", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "groq/meta-llama/llama-4-maverick-17b-128e-instruct", "name": "groq/meta-llama/llama-4-maverick-17b-128e-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-03-09", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "groq/meta-llama/llama-4-scout-17b-16e-instruct", "name": "groq/meta-llama/llama-4-scout-17b-16e-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "output": [ { "amount": 0.33999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-07-17", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "groq/moonshotai/kimi-k2-instruct-0905", "name": "groq/moonshotai/kimi-k2-instruct-0905", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 262144, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-04-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "groq/openai/gpt-oss-120b", "name": "groq/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "groq/openai/gpt-oss-20b", "name": "groq/openai/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.0375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "groq/openai/gpt-oss-safeguard-20b", "name": "groq/openai/gpt-oss-safeguard-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.037, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.7e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "groq/canopylabs/orpheus-v1-english", "name": "groq/canopylabs/orpheus-v1-english", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000022, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.000022" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "groq", "maxInputTokens": 4000, "maxOutputTokens": 50000, "maxTokens": 50000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "groq/canopylabs/orpheus-arabic-saudi", "name": "groq/canopylabs/orpheus-arabic-saudi", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00004, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00004" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "groq", "maxInputTokens": 4000, "maxOutputTokens": 50000, "maxTokens": 50000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "groq/playai-tts", "name": "groq/playai-tts", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00005, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00005" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "groq", "maxInputTokens": 10000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": "2025-12-31", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "groq/qwen/qwen3.6-27b", "name": "groq/qwen/qwen3.6-27b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 131072, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "groq/qwen/qwen3-32b", "name": "groq/qwen/qwen3-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.9e-7" } ], "output": [ { "amount": 0.59, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "groq", "maxInputTokens": 131000, "maxOutputTokens": 131000, "maxTokens": 131000, "deprecationDate": "2026-07-17", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "groq/whisper-large-v3", "name": "groq/whisper-large-v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00003083, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00003083" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "groq", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "groq/whisper-large-v3-turbo", "name": "groq/whisper-large-v3-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00001111, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00001111" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "groq", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "hd/1024-x-1024/dall-e-3", "name": "hd/1024-x-1024/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 7.629e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "7.629e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "hd/1024-x-1792/dall-e-3", "name": "hd/1024-x-1792/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 6.539e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "6.539e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "hd/1792-x-1024/dall-e-3", "name": "hd/1792-x-1024/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 6.539e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "6.539e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "high/1024-x-1024/gpt-image-1", "name": "high/1024-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.167, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.167" }, { "amount": 1.59263611e-7, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.59263611e-7" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "high/1024-x-1536/gpt-image-1", "name": "high/1024-x-1536/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.25, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.25" }, { "amount": 1.58945719e-7, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.58945719e-7" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "high/1536-x-1024/gpt-image-1", "name": "high/1536-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.25, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.25" }, { "amount": 1.58945719e-7, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.58945719e-7" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "hyperbolic/NousResearch/Hermes-3-Llama-3.1-70B", "name": "hyperbolic/NousResearch/Hermes-3-Llama-3.1-70B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/Qwen/QwQ-32B", "name": "hyperbolic/Qwen/QwQ-32B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/Qwen/Qwen2.5-72B-Instruct", "name": "hyperbolic/Qwen/Qwen2.5-72B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/Qwen/Qwen2.5-Coder-32B-Instruct", "name": "hyperbolic/Qwen/Qwen2.5-Coder-32B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/Qwen/Qwen3-235B-A22B", "name": "hyperbolic/Qwen/Qwen3-235B-A22B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/deepseek-ai/DeepSeek-R1", "name": "hyperbolic/deepseek-ai/DeepSeek-R1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/deepseek-ai/DeepSeek-R1-0528", "name": "hyperbolic/deepseek-ai/DeepSeek-R1-0528", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/deepseek-ai/DeepSeek-V3", "name": "hyperbolic/deepseek-ai/DeepSeek-V3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/deepseek-ai/DeepSeek-V3-0324", "name": "hyperbolic/deepseek-ai/DeepSeek-V3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/meta-llama/Llama-3.2-3B-Instruct", "name": "hyperbolic/meta-llama/Llama-3.2-3B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/meta-llama/Llama-3.3-70B-Instruct", "name": "hyperbolic/meta-llama/Llama-3.3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/meta-llama/Meta-Llama-3-70B-Instruct", "name": "hyperbolic/meta-llama/Meta-Llama-3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/meta-llama/Meta-Llama-3.1-405B-Instruct", "name": "hyperbolic/meta-llama/Meta-Llama-3.1-405B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/meta-llama/Meta-Llama-3.1-70B-Instruct", "name": "hyperbolic/meta-llama/Meta-Llama-3.1-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/meta-llama/Meta-Llama-3.1-8B-Instruct", "name": "hyperbolic/meta-llama/Meta-Llama-3.1-8B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "hyperbolic/moonshotai/Kimi-K2-Instruct", "name": "hyperbolic/moonshotai/Kimi-K2-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "hyperbolic", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "j2-light", "name": "j2-light", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "ai21", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "j2-mid", "name": "j2-mid", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "ai21", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "j2-ultra", "name": "j2-ultra", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "ai21", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "jamba-1.5", "name": "jamba-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ai21", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "jamba-1.5-large", "name": "jamba-1.5-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ai21", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "jamba-1.5-large@001", "name": "jamba-1.5-large@001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ai21", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "jamba-1.5-mini", "name": "jamba-1.5-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ai21", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "jamba-1.5-mini@001", "name": "jamba-1.5-mini@001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ai21", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "jamba-large-1.6", "name": "jamba-large-1.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ai21", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "jamba-large-1.7", "name": "jamba-large-1.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ai21", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "jamba-mini-1.6", "name": "jamba-mini-1.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ai21", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "jamba-mini-1.7", "name": "jamba-mini-1.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ai21", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "jina-reranker-v2-base-multilingual", "name": "jina-reranker-v2-base-multilingual", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.018, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "1.8e-8" } ], "output": [ { "amount": 0.018, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "1.8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "jina_ai", "maxInputTokens": 1024, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "jp.anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "jp.anthropic.claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" }, { "amount": 24.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002475" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" }, { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" }, { "amount": 8.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000825" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" }, { "amount": 13.200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000132" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "jp.anthropic.claude-haiku-4-5-20251001-v1:0", "name": "jp.anthropic.claude-haiku-4-5-20251001-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" }, { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "crusoe/deepseek-ai/DeepSeek-R1-0528", "name": "crusoe/deepseek-ai/DeepSeek-R1-0528", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000007" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "crusoe", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "crusoe/deepseek-ai/DeepSeek-V3-0324", "name": "crusoe/deepseek-ai/DeepSeek-V3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "crusoe", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "crusoe/google/gemma-3-12b-it", "name": "crusoe/google/gemma-3-12b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "crusoe", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "crusoe/meta-llama/Llama-3.3-70B-Instruct", "name": "crusoe/meta-llama/Llama-3.3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "crusoe", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "crusoe/moonshotai/Kimi-K2-Thinking", "name": "crusoe/moonshotai/Kimi-K2-Thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "crusoe", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "crusoe/openai/gpt-oss-120b", "name": "crusoe/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "crusoe", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "crusoe/Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "crusoe/Qwen/Qwen3-235B-A22B-Instruct-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "crusoe", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "inception/mercury-2", "name": "inception/mercury-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "inception", "maxInputTokens": 128000, "maxOutputTokens": 50000, "maxTokens": 50000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "text-completion-inception/mercury-edit-2", "name": "text-completion-inception/mercury-edit-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "text-completion-inception", "maxInputTokens": 32000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "lambda_ai/deepseek-llama3.3-70b", "name": "lambda_ai/deepseek-llama3.3-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/deepseek-r1-0528", "name": "lambda_ai/deepseek-r1-0528", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/deepseek-r1-671b", "name": "lambda_ai/deepseek-r1-671b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/deepseek-v3-0324", "name": "lambda_ai/deepseek-v3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/hermes3-405b", "name": "lambda_ai/hermes3-405b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/hermes3-70b", "name": "lambda_ai/hermes3-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/hermes3-8b", "name": "lambda_ai/hermes3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/lfm-40b", "name": "lambda_ai/lfm-40b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/lfm-7b", "name": "lambda_ai/lfm-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/llama-4-maverick-17b-128e-instruct-fp8", "name": "lambda_ai/llama-4-maverick-17b-128e-instruct-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/llama-4-scout-17b-16e-instruct", "name": "lambda_ai/llama-4-scout-17b-16e-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 16384, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/llama3.1-405b-instruct-fp8", "name": "lambda_ai/llama3.1-405b-instruct-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/llama3.1-70b-instruct-fp8", "name": "lambda_ai/llama3.1-70b-instruct-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/llama3.1-8b-instruct", "name": "lambda_ai/llama3.1-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/llama3.1-nemotron-70b-instruct-fp8", "name": "lambda_ai/llama3.1-nemotron-70b-instruct-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/llama3.2-11b-vision-instruct", "name": "lambda_ai/llama3.2-11b-vision-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-8" } ], "output": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/llama3.2-3b-instruct", "name": "lambda_ai/llama3.2-3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.015, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-8" } ], "output": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/llama3.3-70b-instruct-fp8", "name": "lambda_ai/llama3.3-70b-instruct-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/qwen25-coder-32b-instruct", "name": "lambda_ai/qwen25-coder-32b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "lambda_ai/qwen3-32b-fp8", "name": "lambda_ai/qwen3-32b-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "lambda_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "low/1024-x-1024/gpt-image-1", "name": "low/1024-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.011, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.011" }, { "amount": 1.0490417e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.0490417e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "low/1024-x-1536/gpt-image-1", "name": "low/1024-x-1536/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.016, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.016" }, { "amount": 1.0172526e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.0172526e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "low/1536-x-1024/gpt-image-1", "name": "low/1536-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.016, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.016" }, { "amount": 1.0172526e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.0172526e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "max-x-max/50-steps/stability.stable-diffusion-xl-v0", "name": "max-x-max/50-steps/stability.stable-diffusion-xl-v0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.036, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.036" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "max-x-max/max-steps/stability.stable-diffusion-xl-v0", "name": "max-x-max/max-steps/stability.stable-diffusion-xl-v0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.072, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.072" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1024-x-1024/gpt-image-1", "name": "medium/1024-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.042, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.042" }, { "amount": 4.0054321e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "4.0054321e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1024-x-1536/gpt-image-1", "name": "medium/1024-x-1536/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.063, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.063" }, { "amount": 4.0054321e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "4.0054321e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1536-x-1024/gpt-image-1", "name": "medium/1536-x-1024/gpt-image-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.063, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.063" }, { "amount": 4.0054321e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "4.0054321e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "low/1024-x-1024/gpt-image-1-mini", "name": "low/1024-x-1024/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "low/1024-x-1536/gpt-image-1-mini", "name": "low/1024-x-1536/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.006, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.006" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "low/1536-x-1024/gpt-image-1-mini", "name": "low/1536-x-1024/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.006, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.006" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1024-x-1024/gpt-image-1-mini", "name": "medium/1024-x-1024/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.011, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.011" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1024-x-1536/gpt-image-1-mini", "name": "medium/1024-x-1536/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.015, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.015" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medium/1536-x-1024/gpt-image-1-mini", "name": "medium/1536-x-1024/gpt-image-1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.015, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.015" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medlm-large", "name": "medlm-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000005, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.000005" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 8192, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "medlm-medium", "name": "medlm-medium", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 5e-7, "currency": "USD", "units": 1, "pricingType": "character", "raw": "5e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 32768, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama2-13b-chat-v1", "name": "meta.llama2-13b-chat-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama2-70b-chat-v1", "name": "meta.llama2-70b-chat-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000195" } ], "output": [ { "amount": 2.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000256" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama3-1-405b-instruct-v1:0", "name": "meta.llama3-1-405b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000532" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama3-1-70b-instruct-v1:0", "name": "meta.llama3-1-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.9900000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.9e-7" } ], "output": [ { "amount": 0.9900000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama3-1-8b-instruct-v1:0", "name": "meta.llama3-1-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama3-2-11b-instruct-v1:0", "name": "meta.llama3-2-11b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama3-2-1b-instruct-v1:0", "name": "meta.llama3-2-1b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama3-2-3b-instruct-v1:0", "name": "meta.llama3-2-3b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama3-2-90b-instruct-v1:0", "name": "meta.llama3-2-90b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama3-3-70b-instruct-v1:0", "name": "meta.llama3-3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama3-70b-instruct-v1:0", "name": "meta.llama3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000265" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama3-8b-instruct-v1:0", "name": "meta.llama3-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama4-maverick-17b-instruct-v1:0", "name": "meta.llama4-maverick-17b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" }, { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.9700000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.7e-7" }, { "amount": 0.48500000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.85e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta.llama4-scout-17b-instruct-v1:0", "name": "meta.llama4-scout-17b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.7e-7" }, { "amount": 0.08499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.5e-8" } ], "output": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" }, { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "meta/muse-spark-1.1", "name": "meta/muse-spark-1.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 4.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000425" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "meta", "maxInputTokens": 1048576, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "meta/muse-spark-1.2", "name": "meta/muse-spark-1.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 4.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000425" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "meta", "maxInputTokens": 1048576, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "meta/muse-spark-1.2-contributor", "name": "meta/muse-spark-1.2-contributor", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [ { "amount": 0.002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-9" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "meta", "maxInputTokens": 1048576, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "minimax.minimax-m2", "name": "minimax.minimax-m2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "minimax.minimax-m2.1", "name": "minimax.minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 196000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "minimax.minimax-m2.5", "name": "minimax.minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "minimax/speech-02-hd", "name": "minimax/speech-02-hd", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.0001" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "minimax", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "minimax/speech-02-turbo", "name": "minimax/speech-02-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00006, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00006" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "minimax", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "minimax/speech-2.6-hd", "name": "minimax/speech-2.6-hd", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.0001" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "minimax", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "minimax/speech-2.6-turbo", "name": "minimax/speech-2.6-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00006, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00006" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "minimax", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "minimax/MiniMax-M2.1", "name": "minimax/MiniMax-M2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "minimax", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "minimax/MiniMax-M2.1-lightning", "name": "minimax/MiniMax-M2.1-lightning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "minimax", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "minimax/MiniMax-M2.5", "name": "minimax/MiniMax-M2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "minimax", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "minimax/MiniMax-M2.5-lightning", "name": "minimax/MiniMax-M2.5-lightning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "minimax", "maxInputTokens": 1000000, "maxOutputTokens": 8192, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "minimax/MiniMax-M2", "name": "minimax/MiniMax-M2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "minimax", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "minimax/MiniMax-M3", "name": "minimax/MiniMax-M3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "minimax", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "mistral.devstral-2-123b", "name": "mistral.devstral-2-123b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 256000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "mistral.magistral-small-2509", "name": "mistral.magistral-small-2509", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "mistral.ministral-3-14b-instruct", "name": "mistral.ministral-3-14b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "mistral.ministral-3-3b-instruct", "name": "mistral.ministral-3-3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "mistral.ministral-3-8b-instruct", "name": "mistral.ministral-3-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "mistral.mistral-7b-instruct-v0:2", "name": "mistral.mistral-7b-instruct-v0:2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral.mistral-large-2402-v1:0", "name": "mistral.mistral-large-2402-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral.mistral-large-2407-v1:0", "name": "mistral.mistral-large-2407-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral.mistral-large-3-675b-instruct", "name": "mistral.mistral-large-3-675b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "mistral.mistral-small-2402-v1:0", "name": "mistral.mistral-small-2402-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral.mixtral-8x7b-instruct-v0:1", "name": "mistral.mixtral-8x7b-instruct-v0:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral.voxtral-mini-3b-2507", "name": "mistral.voxtral-mini-3b-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "mistral.voxtral-small-24b-2507", "name": "mistral.voxtral-small-24b-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "mistral/codestral-2405", "name": "mistral/codestral-2405", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": "2025-06-16", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/codestral-2508", "name": "mistral/codestral-2508", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/codestral-latest", "name": "mistral/codestral-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/codestral-mamba-latest", "name": "mistral/codestral-mamba-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/devstral-medium-2507", "name": "mistral/devstral-medium-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-05-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/devstral-small-2505", "name": "mistral/devstral-small-2505", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2025-11-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/devstral-small-2507", "name": "mistral/devstral-small-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-05-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/devstral-small-latest", "name": "mistral/devstral-small-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/labs-devstral-small-2512", "name": "mistral/labs-devstral-small-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": "2026-03-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/devstral-latest", "name": "mistral/devstral-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/devstral-medium-latest", "name": "mistral/devstral-medium-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/devstral-2512", "name": "mistral/devstral-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": "2026-07-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/magistral-medium-2506", "name": "mistral/magistral-medium-2506", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 40000, "maxOutputTokens": 40000, "maxTokens": 40000, "deprecationDate": "2025-11-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/magistral-medium-2509", "name": "mistral/magistral-medium-2509", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 40000, "maxOutputTokens": 40000, "maxTokens": 40000, "deprecationDate": "2026-07-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/magistral-medium-1-2-2509", "name": "mistral/magistral-medium-1-2-2509", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 40000, "maxOutputTokens": 40000, "maxTokens": 40000, "deprecationDate": "2026-07-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-ocr-latest", "name": "mistral/mistral-ocr-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.004, "currency": "USD", "units": 1, "pricingType": "page", "raw": "0.004" } ] } ], "metadata": { "source": "litellm", "mode": "ocr", "servingProvider": "mistral", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-ocr-4-0", "name": "mistral/mistral-ocr-4-0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.004, "currency": "USD", "units": 1, "pricingType": "page", "raw": "0.004" } ] } ], "metadata": { "source": "litellm", "mode": "ocr", "servingProvider": "mistral", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-ocr-2505-completion", "name": "mistral/mistral-ocr-2505-completion", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.001, "currency": "USD", "units": 1, "pricingType": "page", "raw": "0.001" } ] } ], "metadata": { "source": "litellm", "mode": "ocr", "servingProvider": "mistral", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-05-31", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-ocr-2512", "name": "mistral/mistral-ocr-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "page", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "ocr", "servingProvider": "mistral", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/magistral-medium-latest", "name": "mistral/magistral-medium-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 40000, "maxOutputTokens": 40000, "maxTokens": 40000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/magistral-small-2506", "name": "mistral/magistral-small-2506", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 40000, "maxOutputTokens": 40000, "maxTokens": 40000, "deprecationDate": "2025-11-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/magistral-small-latest", "name": "mistral/magistral-small-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 40000, "maxOutputTokens": 40000, "maxTokens": 40000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/magistral-small-1-2-2509", "name": "mistral/magistral-small-1-2-2509", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 40000, "maxOutputTokens": 40000, "maxTokens": 40000, "deprecationDate": "2026-07-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-embed", "name": "mistral/mistral-embed", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "mistral", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/codestral-embed", "name": "mistral/codestral-embed", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.5e-7" } ], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "mistral", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/codestral-embed-2505", "name": "mistral/codestral-embed-2505", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.5e-7" } ], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "mistral", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-large-2402", "name": "mistral/mistral-large-2402", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": "2025-06-16", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-large-2407", "name": "mistral/mistral-large-2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2025-03-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-large-2411", "name": "mistral/mistral-large-2411", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-05-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-large-latest", "name": "mistral/mistral-large-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-large-3", "name": "mistral/mistral-large-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-large-2512", "name": "mistral/mistral-large-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-medium", "name": "mistral/mistral-medium", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "output": [ { "amount": 8.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000081" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-medium-2312", "name": "mistral/mistral-medium-2312", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "output": [ { "amount": 8.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000081" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": "2025-06-16", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-medium-2505", "name": "mistral/mistral-medium-2505", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 131072, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": "2026-08-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-medium-2508", "name": "mistral/mistral-medium-2508", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2026-08-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-medium-2604", "name": "mistral/mistral-medium-2604", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-medium-latest", "name": "mistral/mistral-medium-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-medium-3-1-2508", "name": "mistral/mistral-medium-3-1-2508", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2026-08-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-medium-3-5", "name": "mistral/mistral-medium-3-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-small", "name": "mistral/mistral-small", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-small-latest", "name": "mistral/mistral-small-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-small-3-2-2506", "name": "mistral/mistral-small-3-2-2506", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2026-07-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/ministral-3-3b-2512", "name": "mistral/ministral-3-3b-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/ministral-3-8b-2512", "name": "mistral/ministral-3-8b-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/ministral-3-14b-2512", "name": "mistral/ministral-3-14b-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/ministral-8b-2512", "name": "mistral/ministral-8b-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/ministral-8b-latest", "name": "mistral/ministral-8b-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-tiny", "name": "mistral/mistral-tiny", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/open-codestral-mamba", "name": "mistral/open-codestral-mamba", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": "2025-06-06", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/open-mistral-7b", "name": "mistral/open-mistral-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": "2025-03-30", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/open-mistral-nemo", "name": "mistral/open-mistral-nemo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/open-mistral-nemo-2407", "name": "mistral/open-mistral-nemo-2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-07-31", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/open-mixtral-8x22b", "name": "mistral/open-mixtral-8x22b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 65336, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": "2025-03-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/open-mixtral-8x7b", "name": "mistral/open-mixtral-8x7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": "2025-03-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/pixtral-12b-2409", "name": "mistral/pixtral-12b-2409", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2025-12-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/pixtral-large-2411", "name": "mistral/pixtral-large-2411", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": "2026-05-31", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/pixtral-large-latest", "name": "mistral/pixtral-large-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot.kimi-k2-thinking", "name": "moonshot.kimi-k2-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "moonshotai.kimi-k2.5", "name": "moonshotai.kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "moonshot/kimi-k2-0711-preview", "name": "moonshot/kimi-k2-0711-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2026-05-25", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "moonshot/kimi-k2-0905-preview", "name": "moonshot/kimi-k2-0905-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": "2026-05-25", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "moonshot/kimi-k2-turbo-preview", "name": "moonshot/kimi-k2-turbo-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000115" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": "2026-05-25", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "moonshot/kimi-k2.5", "name": "moonshot/kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/kimi-k2.6", "name": "moonshot/kimi-k2.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.5e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/kimi-latest", "name": "moonshot/kimi-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2026-01-28", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/kimi-latest-128k", "name": "moonshot/kimi-latest-128k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2026-01-28", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/kimi-latest-32k", "name": "moonshot/kimi-latest-32k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-01-28", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/kimi-latest-8k", "name": "moonshot/kimi-latest-8k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-01-28", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/kimi-thinking-preview", "name": "moonshot/kimi-thinking-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2025-11-11", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/kimi-k2-thinking", "name": "moonshot/kimi-k2-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": "2026-05-25", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "moonshot/kimi-k2-thinking-turbo", "name": "moonshot/kimi-k2-thinking-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000115" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": "2026-05-25", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "moonshot/moonshot-v1-128k", "name": "moonshot/moonshot-v1-128k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/moonshot-v1-128k-0430", "name": "moonshot/moonshot-v1-128k-0430", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2024-04-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/moonshot-v1-128k-vision-preview", "name": "moonshot/moonshot-v1-128k-vision-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/moonshot-v1-32k", "name": "moonshot/moonshot-v1-32k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/moonshot-v1-32k-0430", "name": "moonshot/moonshot-v1-32k-0430", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2024-04-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/moonshot-v1-32k-vision-preview", "name": "moonshot/moonshot-v1-32k-vision-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/moonshot-v1-8k", "name": "moonshot/moonshot-v1-8k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/moonshot-v1-8k-0430", "name": "moonshot/moonshot-v1-8k-0430", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2024-04-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/moonshot-v1-8k-vision-preview", "name": "moonshot/moonshot-v1-8k-vision-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "moonshot/moonshot-v1-auto", "name": "moonshot/moonshot-v1-auto", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "moonshot", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "morph/morph-v3-fast", "name": "morph/morph-v3-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "morph", "maxInputTokens": 16000, "maxOutputTokens": 16000, "maxTokens": 16000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "morph/morph-v3-large", "name": "morph/morph-v3-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 1.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000019" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "morph", "maxInputTokens": 16000, "maxOutputTokens": 16000, "maxTokens": 16000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "multimodalembedding", "name": "multimodalembedding", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "8e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0005, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.0005" }, { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0001" }, { "amount": 2e-7, "currency": "USD", "units": 1, "pricingType": "character", "raw": "2e-7" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai-embedding-models", "maxInputTokens": 2048, "maxOutputTokens": null, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "multimodalembedding@001", "name": "multimodalembedding@001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "8e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0005, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.0005" }, { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0001" }, { "amount": 2e-7, "currency": "USD", "units": 1, "pricingType": "character", "raw": "2e-7" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai-embedding-models", "maxInputTokens": 2048, "maxOutputTokens": null, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/Qwen/QwQ-32B", "name": "nscale/Qwen/QwQ-32B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/Qwen/Qwen2.5-Coder-32B-Instruct", "name": "nscale/Qwen/Qwen2.5-Coder-32B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/Qwen/Qwen2.5-Coder-3B-Instruct", "name": "nscale/Qwen/Qwen2.5-Coder-3B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-8" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/Qwen/Qwen2.5-Coder-7B-Instruct", "name": "nscale/Qwen/Qwen2.5-Coder-7B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-8" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/black-forest-labs/FLUX.1-schnell", "name": "nscale/black-forest-labs/FLUX.1-schnell", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 1.3e-9, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "1.3e-9" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/deepseek-ai/DeepSeek-R1-Distill-Llama-70B", "name": "nscale/deepseek-ai/DeepSeek-R1-Distill-Llama-70B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" } ], "output": [ { "amount": 0.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.75e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/deepseek-ai/DeepSeek-R1-Distill-Llama-8B", "name": "nscale/deepseek-ai/DeepSeek-R1-Distill-Llama-8B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "output": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B", "name": "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-14B", "name": "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-14B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B", "name": "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-7B", "name": "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-7B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/meta-llama/Llama-3.1-8B-Instruct", "name": "nscale/meta-llama/Llama-3.1-8B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/meta-llama/Llama-3.3-70B-Instruct", "name": "nscale/meta-llama/Llama-3.3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct", "name": "nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/mistralai/mixtral-8x22b-instruct-v0.1", "name": "nscale/mistralai/mixtral-8x22b-instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nscale/stabilityai/stable-diffusion-xl-base-1.0", "name": "nscale/stabilityai/stable-diffusion-xl-base-1.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 3e-9, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "3e-9" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "nscale", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/deepseek-ai/DeepSeek-R1", "name": "nebius/deepseek-ai/DeepSeek-R1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/deepseek-ai/DeepSeek-R1-0528", "name": "nebius/deepseek-ai/DeepSeek-R1-0528", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 164000, "maxOutputTokens": 164000, "maxTokens": 164000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/deepseek-ai/DeepSeek-R1-Distill-Llama-70B", "name": "nebius/deepseek-ai/DeepSeek-R1-Distill-Llama-70B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/deepseek-ai/DeepSeek-V3", "name": "nebius/deepseek-ai/DeepSeek-V3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/deepseek-ai/DeepSeek-V3-0324", "name": "nebius/deepseek-ai/DeepSeek-V3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/google/gemma-3-27b-it", "name": "nebius/google/gemma-3-27b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/meta-llama/Llama-3.3-70B-Instruct", "name": "nebius/meta-llama/Llama-3.3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/meta-llama/Llama-Guard-3-8B", "name": "nebius/meta-llama/Llama-Guard-3-8B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "output": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/meta-llama/Meta-Llama-3.1-8B-Instruct", "name": "nebius/meta-llama/Meta-Llama-3.1-8B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "output": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/meta-llama/Meta-Llama-3.1-70B-Instruct", "name": "nebius/meta-llama/Meta-Llama-3.1-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/meta-llama/Meta-Llama-3.1-405B-Instruct", "name": "nebius/meta-llama/Meta-Llama-3.1-405B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/mistralai/Mistral-Nemo-Instruct-2407", "name": "nebius/mistralai/Mistral-Nemo-Instruct-2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/NousResearch/Hermes-3-Llama-3.1-405B", "name": "nebius/NousResearch/Hermes-3-Llama-3.1-405B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/nvidia/Llama-3.1-Nemotron-Ultra-253B-v1", "name": "nebius/nvidia/Llama-3.1-Nemotron-Ultra-253B-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/nvidia/Llama-3.3-Nemotron-Super-49B-v1", "name": "nebius/nvidia/Llama-3.3-Nemotron-Super-49B-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/Qwen3-235B-A22B", "name": "nebius/Qwen/Qwen3-235B-A22B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/Qwen3-32B", "name": "nebius/Qwen/Qwen3-32B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/Qwen3-30B-A3B", "name": "nebius/Qwen/Qwen3-30B-A3B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/Qwen3-14B", "name": "nebius/Qwen/Qwen3-14B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/Qwen3-4B", "name": "nebius/Qwen/Qwen3-4B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/QwQ-32B", "name": "nebius/Qwen/QwQ-32B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/Qwen2.5-72B-Instruct", "name": "nebius/Qwen/Qwen2.5-72B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/Qwen2.5-32B-Instruct", "name": "nebius/Qwen/Qwen2.5-32B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/Qwen2.5-Coder-7B", "name": "nebius/Qwen/Qwen2.5-Coder-7B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-8" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/Qwen2.5-VL-72B-Instruct", "name": "nebius/Qwen/Qwen2.5-VL-72B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/Qwen2-VL-72B-Instruct", "name": "nebius/Qwen/Qwen2-VL-72B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/Qwen/Qwen2-VL-7B-Instruct", "name": "nebius/Qwen/Qwen2-VL-7B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "output": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "nebius", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/BAAI/bge-en-icl", "name": "nebius/BAAI/bge-en-icl", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "nebius", "maxInputTokens": 32768, "maxOutputTokens": null, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/BAAI/bge-multilingual-gemma2", "name": "nebius/BAAI/bge-multilingual-gemma2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "nebius", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nebius/intfloat/e5-mistral-7b-instruct", "name": "nebius/intfloat/e5-mistral-7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "nebius", "maxInputTokens": 32768, "maxOutputTokens": null, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nvidia.nemotron-nano-12b-v2", "name": "nvidia.nemotron-nano-12b-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "nvidia.nemotron-nano-9b-v2", "name": "nvidia.nemotron-nano-9b-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.22999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "nvidia.nemotron-nano-3-30b", "name": "nvidia.nemotron-nano-3-30b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "nvidia.nemotron-super-3-120b", "name": "nvidia.nemotron-super-3-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 256000, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "o1", "name": "o1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "o1-2024-12-17", "name": "o1-2024-12-17", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "o1-pro", "name": "o1-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 150, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00015" }, { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "output": [ { "amount": 600, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0006" }, { "amount": 300, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "o1-pro-2025-03-19", "name": "o1-pro-2025-03-19", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 150, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00015" }, { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "output": [ { "amount": 600, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0006" }, { "amount": 300, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "o3", "name": "o3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" }, { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "o3-2025-04-16", "name": "o3-2025-04-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" }, { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 0.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-12-11", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "o3-deep-research", "name": "o3-deep-research", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00004" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "o3-deep-research-2025-06-26", "name": "o3-deep-research-2025-06-26", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00004" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "cacheRead": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "o3-mini", "name": "o3-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "o3-mini-2025-01-31", "name": "o3-mini-2025-01-31", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "o3-pro", "name": "o3-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00008" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "o3-pro-2025-06-10", "name": "o3-pro-2025-06-10", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 80, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00008" }, { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-12-11", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "o4-mini", "name": "o4-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "o4-mini-2025-04-16", "name": "o4-mini-2025-04-16", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" }, { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-10-23", "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "o4-mini-deep-research", "name": "o4-mini-deep-research", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "o4-mini-deep-research-2025-06-26", "name": "o4-mini-deep-research-2025-06-26", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "openai", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "oci/meta.llama-3.1-8b-instruct", "name": "oci/meta.llama-3.1-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/meta.llama-3.1-70b-instruct", "name": "oci/meta.llama-3.1-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/meta.llama-3.1-405b-instruct", "name": "oci/meta.llama-3.1-405b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001068" } ], "output": [ { "amount": 10.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001068" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/meta.llama-3.2-90b-vision-instruct", "name": "oci/meta.llama-3.2-90b-vision-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/meta.llama-3.3-70b-instruct", "name": "oci/meta.llama-3.3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/meta.llama-4-maverick-17b-128e-instruct-fp8", "name": "oci/meta.llama-4-maverick-17b-128e-instruct-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/meta.llama-4-scout-17b-16e-instruct", "name": "oci/meta.llama-4-scout-17b-16e-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 10485760, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/xai.grok-3", "name": "oci/xai.grok-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/xai.grok-3-fast", "name": "oci/xai.grok-3-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/xai.grok-3-mini", "name": "oci/xai.grok-3-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/xai.grok-3-mini-fast", "name": "oci/xai.grok-3-mini-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/xai.grok-4", "name": "oci/xai.grok-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.command-latest", "name": "oci/cohere.command-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "output": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.command-a-03-2025", "name": "oci/cohere.command-a-03-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "output": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 256000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.command-plus-latest", "name": "oci/cohere.command-plus-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "output": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/google.gemini-2.5-flash", "name": "oci/google.gemini-2.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "oci/google.gemini-2.5-pro", "name": "oci/google.gemini-2.5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "oci/google.gemini-2.5-flash-lite", "name": "oci/google.gemini-2.5-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.command-a-vision", "name": "oci/cohere.command-a-vision", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "output": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 256000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.command-a-reasoning", "name": "oci/cohere.command-a-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "output": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 256000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.embed-multilingual-image-v3.0", "name": "oci/cohere.embed-multilingual-image-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "oci", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.command-a-reasoning-08-2025", "name": "oci/cohere.command-a-reasoning-08-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "output": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 256000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.command-a-vision-07-2025", "name": "oci/cohere.command-a-vision-07-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "output": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.command-a-translate-08-2025", "name": "oci/cohere.command-a-translate-08-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 256000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.command-r-08-2024", "name": "oci/cohere.command-r-08-2024", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.command-r-plus-08-2024", "name": "oci/cohere.command-r-plus-08-2024", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "output": [ { "amount": 1.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000156" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/meta.llama-3.2-11b-vision-instruct", "name": "oci/meta.llama-3.2-11b-vision-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/meta.llama-3.3-70b-instruct-fp8-dynamic", "name": "oci/meta.llama-3.3-70b-instruct-fp8-dynamic", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/xai.grok-4-fast", "name": "oci/xai.grok-4-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/xai.grok-4.1-fast", "name": "oci/xai.grok-4.1-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/xai.grok-4.20", "name": "oci/xai.grok-4.20", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/xai.grok-4.20-multi-agent", "name": "oci/xai.grok-4.20-multi-agent", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/xai.grok-code-fast-1", "name": "oci/xai.grok-code-fast-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "oci/openai.gpt-5", "name": "oci/openai.gpt-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "oci/openai.gpt-5-mini", "name": "oci/openai.gpt-5-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "oci/openai.gpt-5-nano", "name": "oci/openai.gpt-5-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "oci", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.embed-english-v3.0", "name": "oci/cohere.embed-english-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "oci", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.embed-english-light-v3.0", "name": "oci/cohere.embed-english-light-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "oci", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.embed-multilingual-v3.0", "name": "oci/cohere.embed-multilingual-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "oci", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.embed-multilingual-light-v3.0", "name": "oci/cohere.embed-multilingual-light-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "oci", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.embed-english-image-v3.0", "name": "oci/cohere.embed-english-image-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "oci", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.embed-english-light-image-v3.0", "name": "oci/cohere.embed-english-light-image-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "oci", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.embed-multilingual-light-image-v3.0", "name": "oci/cohere.embed-multilingual-light-image-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "oci", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "oci/cohere.embed-v4.0", "name": "oci/cohere.embed-v4.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "oci", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/codegeex4", "name": "ollama/codegeex4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 32768, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/codegemma", "name": "ollama/codegemma", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "ollama", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/codellama", "name": "ollama/codellama", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "ollama", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/deepseek-coder-v2-base", "name": "ollama/deepseek-coder-v2-base", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "ollama", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/deepseek-coder-v2-instruct", "name": "ollama/deepseek-coder-v2-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 32768, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/deepseek-coder-v2-lite-base", "name": "ollama/deepseek-coder-v2-lite-base", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "ollama", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/deepseek-coder-v2-lite-instruct", "name": "ollama/deepseek-coder-v2-lite-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 32768, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/deepseek-v3.1:671b-cloud", "name": "ollama/deepseek-v3.1:671b-cloud", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/gpt-oss:120b-cloud", "name": "ollama/gpt-oss:120b-cloud", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/gpt-oss:20b-cloud", "name": "ollama/gpt-oss:20b-cloud", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/internlm2_5-20b-chat", "name": "ollama/internlm2_5-20b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 32768, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/llama2", "name": "ollama/llama2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/llama2-uncensored", "name": "ollama/llama2-uncensored", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "ollama", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/llama2:13b", "name": "ollama/llama2:13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/llama2:70b", "name": "ollama/llama2:70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/llama2:7b", "name": "ollama/llama2:7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/llama3", "name": "ollama/llama3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/llama3.1", "name": "ollama/llama3.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/llama3:70b", "name": "ollama/llama3:70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/llama3:8b", "name": "ollama/llama3:8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/mistral", "name": "ollama/mistral", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "ollama", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/mistral-7B-Instruct-v0.1", "name": "ollama/mistral-7B-Instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/mistral-7B-Instruct-v0.2", "name": "ollama/mistral-7B-Instruct-v0.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/mistral-large-instruct-2407", "name": "ollama/mistral-large-instruct-2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 65536, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/mixtral-8x22B-Instruct-v0.1", "name": "ollama/mixtral-8x22B-Instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 65536, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/mixtral-8x7B-Instruct-v0.1", "name": "ollama/mixtral-8x7B-Instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/orca-mini", "name": "ollama/orca-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "ollama", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/qwen3-coder:480b-cloud", "name": "ollama/qwen3-coder:480b-cloud", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ollama", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ollama/vicuna", "name": "ollama/vicuna", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "ollama", "maxInputTokens": 2048, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "omni-moderation-2024-09-26", "name": "omni-moderation-2024-09-26", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "moderation", "servingProvider": "openai", "maxInputTokens": 32768, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "omni-moderation-latest", "name": "omni-moderation-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "moderation", "servingProvider": "openai", "maxInputTokens": 32768, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openai.gpt-oss-120b-1:0", "name": "openai.gpt-oss-120b-1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "openai.gpt-oss-20b-1:0", "name": "openai.gpt-oss-20b-1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "openai.gpt-oss-safeguard-120b", "name": "openai.gpt-oss-safeguard-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "openai.gpt-oss-safeguard-20b", "name": "openai.gpt-oss-safeguard-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-3-haiku", "name": "openrouter/anthropic/claude-3-haiku", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0004, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0004" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-3.5-sonnet", "name": "openrouter/anthropic/claude-3.5-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-3.7-sonnet", "name": "openrouter/anthropic/claude-3.7-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0048, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0048" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-opus-4", "name": "openrouter/anthropic/claude-opus-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [ { "amount": 0.0048, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0048" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-opus-4.1", "name": "openrouter/anthropic/claude-opus-4.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "other": [ { "amount": 0.0048, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0048" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-sonnet-4", "name": "openrouter/anthropic/claude-sonnet-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "other": [ { "amount": 0.0048, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0048" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-sonnet-4.6", "name": "openrouter/anthropic/claude-sonnet-4.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-opus-4.5", "name": "openrouter/anthropic/claude-opus-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-opus-4.6", "name": "openrouter/anthropic/claude-opus-4.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-sonnet-4.5", "name": "openrouter/anthropic/claude-sonnet-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "other": [ { "amount": 0.0048, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0048" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-haiku-4.5", "name": "openrouter/anthropic/claude-haiku-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 200000, "maxOutputTokens": 200000, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/anthropic/claude-opus-4.7", "name": "openrouter/anthropic/claude-opus-4.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/bytedance/ui-tars-1.5-7b", "name": "openrouter/bytedance/ui-tars-1.5-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 131072, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/deepseek/deepseek-chat", "name": "openrouter/deepseek/deepseek-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 65536, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/deepseek/deepseek-chat-v3-0324", "name": "openrouter/deepseek/deepseek-chat-v3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 65536, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/deepseek/deepseek-chat-v3.1", "name": "openrouter/deepseek/deepseek-chat-v3.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/deepseek/deepseek-v3.2", "name": "openrouter/deepseek/deepseek-v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/deepseek/deepseek-v3.2-exp", "name": "openrouter/deepseek/deepseek-v3.2-exp", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": false, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/deepseek/deepseek-r1", "name": "openrouter/deepseek/deepseek-r1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "output": [ { "amount": 2.1900000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000219" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 65336, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/deepseek/deepseek-r1-0528", "name": "openrouter/deepseek/deepseek-r1-0528", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 2.1500000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000215" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 65336, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/google/gemini-2.0-flash-001", "name": "openrouter/google/gemini-2.0-flash-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "openrouter/google/gemini-2.5-flash", "name": "openrouter/google/gemini-2.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "openrouter/google/gemini-2.5-pro", "name": "openrouter/google/gemini-2.5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7e-7" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "openrouter/google/gemini-3-pro-preview", "name": "openrouter/google/gemini-3-pro-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "openrouter/google/gemini-3-flash-preview", "name": "openrouter/google/gemini-3-flash-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "openrouter/google/gemini-3.1-flash-lite-preview", "name": "openrouter/google/gemini-3.1-flash-lite-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "openrouter/google/gemini-3.1-flash-lite", "name": "openrouter/google/gemini-3.1-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "openrouter/google/gemini-3.1-pro-preview", "name": "openrouter/google/gemini-3.1-pro-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" }, { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "openrouter/gryphe/mythomax-l2-13b", "name": "openrouter/gryphe/mythomax-l2-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "output": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/mancer/weaver", "name": "openrouter/mancer/weaver", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005625" } ], "output": [ { "amount": 5.625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005625" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 8000, "maxOutputTokens": 2000, "maxTokens": 2000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/meta-llama/llama-3-70b-instruct", "name": "openrouter/meta-llama/llama-3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.59, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.9e-7" } ], "output": [ { "amount": 0.7899999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 8192, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/minimax/minimax-m2", "name": "openrouter/minimax/minimax-m2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.255, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.55e-7" } ], "output": [ { "amount": 1.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000102" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 204800, "maxOutputTokens": 204800, "maxTokens": 204800, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/mistralai/devstral-2512", "name": "openrouter/mistralai/devstral-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/mistralai/ministral-3b-2512", "name": "openrouter/mistralai/ministral-3b-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/mistralai/ministral-8b-2512", "name": "openrouter/mistralai/ministral-8b-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/mistralai/ministral-14b-2512", "name": "openrouter/mistralai/ministral-14b-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/mistralai/mistral-large-2512", "name": "openrouter/mistralai/mistral-large-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/mistralai/mistral-7b-instruct", "name": "openrouter/mistralai/mistral-7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 32768, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/mistralai/mistral-large", "name": "openrouter/mistralai/mistral-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "output": [ { "amount": 24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 128000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/mistralai/mistral-small-3.1-24b-instruct", "name": "openrouter/mistralai/mistral-small-3.1-24b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/mistralai/mistral-small-3.2-24b-instruct", "name": "openrouter/mistralai/mistral-small-3.2-24b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/mistralai/mixtral-8x22b-instruct", "name": "openrouter/mistralai/mixtral-8x22b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "output": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 65536, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/moonshotai/kimi-k2.5", "name": "openrouter/moonshotai/kimi-k2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/nvidia/nemotron-3.5-lightning", "name": "openrouter/nvidia/nemotron-3.5-lightning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-3.5-turbo", "name": "openrouter/openai/gpt-3.5-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-3.5-turbo-16k", "name": "openrouter/openai/gpt-3.5-turbo-16k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-4", "name": "openrouter/openai/gpt-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 8191, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-4.1", "name": "openrouter/openai/gpt-4.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "openrouter/openai/gpt-4.1-mini", "name": "openrouter/openai/gpt-4.1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "openrouter/openai/gpt-4.1-nano", "name": "openrouter/openai/gpt-4.1-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "openrouter/openai/gpt-4o", "name": "openrouter/openai/gpt-4o", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-4o-2024-05-13", "name": "openrouter/openai/gpt-4o-2024-05-13", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-5-chat", "name": "openrouter/openai/gpt-5-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-5-codex", "name": "openrouter/openai/gpt-5-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-5.2-codex", "name": "openrouter/openai/gpt-5.2-codex", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-5", "name": "openrouter/openai/gpt-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-5-mini", "name": "openrouter/openai/gpt-5-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-5-nano", "name": "openrouter/openai/gpt-5-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-9" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-5.1-codex-max", "name": "openrouter/openai/gpt-5.1-codex-max", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 400000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-5.2", "name": "openrouter/openai/gpt-5.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" } ], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-5.2-chat", "name": "openrouter/openai/gpt-5.2-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "output": [ { "amount": 14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000014" } ], "cacheRead": [ { "amount": 0.175, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.75e-7" } ], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-5.2-pro", "name": "openrouter/openai/gpt-5.2-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000021" } ], "output": [ { "amount": 168, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000168" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-oss-120b", "name": "openrouter/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/gpt-oss-20b", "name": "openrouter/openai/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/o1", "name": "openrouter/openai/o1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "openrouter/openai/o3-mini", "name": "openrouter/openai/o3-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 128000, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openai/o3-mini-high", "name": "openrouter/openai/o3-mini-high", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 128000, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen-2.5-coder-32b-instruct", "name": "openrouter/qwen/qwen-2.5-coder-32b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 33792, "maxOutputTokens": 33792, "maxTokens": 33792, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen-vl-plus", "name": "openrouter/qwen/qwen-vl-plus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.1e-7" } ], "output": [ { "amount": 0.63, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 8192, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen3-coder", "name": "openrouter/qwen/qwen3-coder", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262100, "maxOutputTokens": 262100, "maxTokens": 262100, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen3-coder-plus", "name": "openrouter/qwen/qwen3-coder-plus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 997952, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen3-235b-a22b-2507", "name": "openrouter/qwen/qwen3-235b-a22b-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.071, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.1e-8" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen3-235b-a22b-thinking-2507", "name": "openrouter/qwen/qwen3-235b-a22b-thinking-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen3.6-plus", "name": "openrouter/qwen/qwen3.6-plus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.325, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.25e-7" } ], "output": [ { "amount": 1.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000195" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1000000, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen3.5-35b-a3b", "name": "openrouter/qwen/qwen3.5-35b-a3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen3.5-27b", "name": "openrouter/qwen/qwen3.5-27b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen3.5-122b-a10b", "name": "openrouter/qwen/qwen3.5-122b-a10b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen3.5-flash-02-23", "name": "openrouter/qwen/qwen3.5-flash-02-23", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1000000, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen3.5-plus-02-15", "name": "openrouter/qwen/qwen3.5-plus-02-15", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1000000, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/qwen/qwen3.5-397b-a17b", "name": "openrouter/qwen/qwen3.5-397b-a17b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/switchpoint/router", "name": "openrouter/switchpoint/router", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.5e-7" } ], "output": [ { "amount": 3.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000034" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/undi95/remm-slerp-l2-13b", "name": "openrouter/undi95/remm-slerp-l2-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "output": [ { "amount": 1.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001875" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 6144, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/x-ai/grok-4", "name": "openrouter/x-ai/grok-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "openrouter/z-ai/glm-4.6", "name": "openrouter/z-ai/glm-4.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 202800, "maxOutputTokens": 131000, "maxTokens": 131000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/z-ai/glm-4.6:exacto", "name": "openrouter/z-ai/glm-4.6:exacto", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 1.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000019" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 202800, "maxOutputTokens": 131000, "maxTokens": 131000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/xiaomi/mimo-v2-flash", "name": "openrouter/xiaomi/mimo-v2-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-8" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 262144, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/xiaomi/mimo-v2.5-pro", "name": "openrouter/xiaomi/mimo-v2.5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1048576, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/xiaomi/mimo-v2.5", "name": "openrouter/xiaomi/mimo-v2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 1048576, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/z-ai/glm-4.7", "name": "openrouter/z-ai/glm-4.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 202752, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/z-ai/glm-4.7-flash", "name": "openrouter/z-ai/glm-4.7-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/z-ai/glm-5", "name": "openrouter/z-ai/glm-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 2.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000256" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 202752, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/z-ai/glm-5.1", "name": "openrouter/z-ai/glm-5.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.0499999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000105" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [ { "amount": 0.5249999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.25e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 202752, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/minimax/minimax-m2.1", "name": "openrouter/minimax/minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 204000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/minimax/minimax-m2.5", "name": "openrouter/minimax/minimax-m2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 196608, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openrouter/auto", "name": "openrouter/openrouter/auto", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 2000000, "maxOutputTokens": null, "maxTokens": 2000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openrouter/free", "name": "openrouter/openrouter/free", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 200000, "maxOutputTokens": null, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "openrouter/openrouter/bodybuilder", "name": "openrouter/openrouter/bodybuilder", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openrouter", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/DeepSeek-R1-Distill-Llama-70B", "name": "ovhcloud/DeepSeek-R1-Distill-Llama-70B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.67, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.7e-7" } ], "output": [ { "amount": 0.67, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 131000, "maxOutputTokens": 131000, "maxTokens": 131000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/Llama-3.1-8B-Instruct", "name": "ovhcloud/Llama-3.1-8B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 131000, "maxOutputTokens": 131000, "maxTokens": 131000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/Meta-Llama-3_1-70B-Instruct", "name": "ovhcloud/Meta-Llama-3_1-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.67, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.7e-7" } ], "output": [ { "amount": 0.67, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 131000, "maxOutputTokens": 131000, "maxTokens": 131000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/Meta-Llama-3_3-70B-Instruct", "name": "ovhcloud/Meta-Llama-3_3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.67, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.7e-7" } ], "output": [ { "amount": 0.67, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 131000, "maxOutputTokens": 131000, "maxTokens": 131000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/Mistral-7B-Instruct-v0.3", "name": "ovhcloud/Mistral-7B-Instruct-v0.3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 127000, "maxOutputTokens": 127000, "maxTokens": 127000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/Mistral-Nemo-Instruct-2407", "name": "ovhcloud/Mistral-Nemo-Instruct-2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 118000, "maxOutputTokens": 118000, "maxTokens": 118000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/Mistral-Small-3.2-24B-Instruct-2506", "name": "ovhcloud/Mistral-Small-3.2-24B-Instruct-2506", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/Mixtral-8x7B-Instruct-v0.1", "name": "ovhcloud/Mixtral-8x7B-Instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.63, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.3e-7" } ], "output": [ { "amount": 0.63, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/Qwen2.5-Coder-32B-Instruct", "name": "ovhcloud/Qwen2.5-Coder-32B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.87, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.7e-7" } ], "output": [ { "amount": 0.87, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/Qwen2.5-VL-72B-Instruct", "name": "ovhcloud/Qwen2.5-VL-72B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.9099999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.1e-7" } ], "output": [ { "amount": 0.9099999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/Qwen3-32B", "name": "ovhcloud/Qwen3-32B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.22999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/gpt-oss-120b", "name": "ovhcloud/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 131000, "maxOutputTokens": 131000, "maxTokens": 131000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/gpt-oss-20b", "name": "ovhcloud/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 131000, "maxOutputTokens": 131000, "maxTokens": 131000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/llava-v1.6-mistral-7b-hf", "name": "ovhcloud/llava-v1.6-mistral-7b-hf", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.9e-7" } ], "output": [ { "amount": 0.29, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "ovhcloud/mamba-codestral-7B-v0.1", "name": "ovhcloud/mamba-codestral-7B-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.9e-7" } ], "output": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "ovhcloud", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "palm/chat-bison", "name": "palm/chat-bison", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "output": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "palm", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "palm/chat-bison-001", "name": "palm/chat-bison-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "output": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "palm", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "palm/text-bison", "name": "palm/text-bison", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "output": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "palm", "maxInputTokens": 8192, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "palm/text-bison-001", "name": "palm/text-bison-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "output": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "palm", "maxInputTokens": 8192, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "palm/text-bison-safety-off", "name": "palm/text-bison-safety-off", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "output": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "palm", "maxInputTokens": 8192, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "palm/text-bison-safety-recitation-off", "name": "palm/text-bison-safety-recitation-off", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "output": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "palm", "maxInputTokens": 8192, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "parallel_ai/search", "name": "parallel_ai/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.004, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.004" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "parallel_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "parallel_ai/search-pro", "name": "parallel_ai/search-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.009, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.009" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "parallel_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/codellama-34b-instruct", "name": "perplexity/codellama-34b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/codellama-70b-instruct", "name": "perplexity/codellama-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "output": [ { "amount": 2.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000028" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/llama-2-70b-chat", "name": "perplexity/llama-2-70b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "output": [ { "amount": 2.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000028" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/llama-3.1-70b-instruct", "name": "perplexity/llama-3.1-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/llama-3.1-8b-instruct", "name": "perplexity/llama-3.1-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/mistral-7b-instruct", "name": "perplexity/mistral-7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/mixtral-8x7b-instruct", "name": "perplexity/mixtral-8x7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/pplx-70b-chat", "name": "perplexity/pplx-70b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "output": [ { "amount": 2.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000028" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/pplx-70b-online", "name": "perplexity/pplx-70b-online", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 2.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000028" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/pplx-7b-chat", "name": "perplexity/pplx-7b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/pplx-7b-online", "name": "perplexity/pplx-7b-online", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/sonar", "name": "perplexity/sonar", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "perplexity/sonar-deep-research", "name": "perplexity/sonar-deep-research", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "perplexity/sonar-medium-chat", "name": "perplexity/sonar-medium-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/sonar-medium-online", "name": "perplexity/sonar-medium-online", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 12000, "maxOutputTokens": 12000, "maxTokens": 12000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/sonar-pro", "name": "perplexity/sonar-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 200000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "perplexity/sonar-reasoning", "name": "perplexity/sonar-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "perplexity/sonar-reasoning-pro", "name": "perplexity/sonar-reasoning-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "perplexity/sonar-small-chat", "name": "perplexity/sonar-small-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/sonar-small-online", "name": "perplexity/sonar-small-online", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "perplexity", "maxInputTokens": 12000, "maxOutputTokens": 12000, "maxTokens": 12000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "publicai/swiss-ai/apertus-8b-instruct", "name": "publicai/swiss-ai/apertus-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "publicai", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "publicai/swiss-ai/apertus-70b-instruct", "name": "publicai/swiss-ai/apertus-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "publicai", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "publicai/aisingapore/Gemma-SEA-LION-v4-27B-IT", "name": "publicai/aisingapore/Gemma-SEA-LION-v4-27B-IT", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "publicai", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "publicai/BSC-LT/salamandra-7b-instruct-tools-16k", "name": "publicai/BSC-LT/salamandra-7b-instruct-tools-16k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "publicai", "maxInputTokens": 16384, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "publicai/BSC-LT/ALIA-40b-instruct_Q8_0", "name": "publicai/BSC-LT/ALIA-40b-instruct_Q8_0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "publicai", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "publicai/allenai/Olmo-3-7B-Instruct", "name": "publicai/allenai/Olmo-3-7B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "publicai", "maxInputTokens": 32768, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/pplx-embed-v1-0.6b", "name": "perplexity/pplx-embed-v1-0.6b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "4e-9" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "perplexity", "maxInputTokens": 32768, "maxOutputTokens": null, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "perplexity/pplx-embed-v1-4b", "name": "perplexity/pplx-embed-v1-4b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "3e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "perplexity", "maxInputTokens": 32768, "maxOutputTokens": null, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "publicai/aisingapore/Qwen-SEA-LION-v4-32B-IT", "name": "publicai/aisingapore/Qwen-SEA-LION-v4-32B-IT", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "publicai", "maxInputTokens": 32768, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "publicai/allenai/Olmo-3-7B-Think", "name": "publicai/allenai/Olmo-3-7B-Think", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "publicai", "maxInputTokens": 32768, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "publicai/allenai/Olmo-3-32B-Think", "name": "publicai/allenai/Olmo-3-32B-Think", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "publicai", "maxInputTokens": 32768, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "qwen.qwen3-coder-480b-a35b-v1:0", "name": "qwen.qwen3-coder-480b-a35b-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 262000, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "qwen.qwen3-235b-a22b-2507-v1:0", "name": "qwen.qwen3-235b-a22b-2507-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 262144, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "qwen.qwen3-coder-30b-a3b-v1:0", "name": "qwen.qwen3-coder-30b-a3b-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 262144, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "qwen.qwen3-32b-v1:0", "name": "qwen.qwen3-32b-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 131072, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "qwen.qwen3-next-80b-a3b", "name": "qwen.qwen3-next-80b-a3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "qwen.qwen3-vl-235b-a22b", "name": "qwen.qwen3-vl-235b-a22b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.53, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.3e-7" } ], "output": [ { "amount": 2.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000266" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "qwen.qwen3-coder-next", "name": "qwen.qwen3-coder-next", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 262144, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "recraft/recraftv2", "name": "recraft/recraftv2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.022, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.022" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "recraft", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "recraft/recraftv3", "name": "recraft/recraftv3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "recraft", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/meta/llama-2-13b", "name": "replicate/meta/llama-2-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/meta/llama-2-13b-chat", "name": "replicate/meta/llama-2-13b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/meta/llama-2-70b", "name": "replicate/meta/llama-2-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "output": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/meta/llama-2-70b-chat", "name": "replicate/meta/llama-2-70b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "output": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/meta/llama-2-7b", "name": "replicate/meta/llama-2-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/meta/llama-2-7b-chat", "name": "replicate/meta/llama-2-7b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/meta/llama-3-70b", "name": "replicate/meta/llama-3-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "output": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/meta/llama-3-70b-instruct", "name": "replicate/meta/llama-3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "output": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/meta/llama-3-8b", "name": "replicate/meta/llama-3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 8086, "maxOutputTokens": 8086, "maxTokens": 8086, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/meta/llama-3-8b-instruct", "name": "replicate/meta/llama-3-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 8086, "maxOutputTokens": 8086, "maxTokens": 8086, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/mistralai/mistral-7b-instruct-v0.2", "name": "replicate/mistralai/mistral-7b-instruct-v0.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/mistralai/mistral-7b-v0.1", "name": "replicate/mistralai/mistral-7b-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/mistralai/mixtral-8x7b-instruct-v0.1", "name": "replicate/mistralai/mixtral-8x7b-instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "replicate/openai/gpt-5", "name": "replicate/openai/gpt-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/gpt-oss-20b", "name": "replicate/openai/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/anthropic/claude-4.5-haiku", "name": "replicate/anthropic/claude-4.5-haiku", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/ibm-granite/granite-3.3-8b-instruct", "name": "replicate/ibm-granite/granite-3.3-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/gpt-4o", "name": "replicate/openai/gpt-4o", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/o4-mini", "name": "replicate/openai/o4-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/o1-mini", "name": "replicate/openai/o1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/o1", "name": "replicate/openai/o1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/gpt-4o-mini", "name": "replicate/openai/gpt-4o-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/qwen/qwen3-235b-a22b-instruct-2507", "name": "replicate/qwen/qwen3-235b-a22b-instruct-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.26399999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.64e-7" } ], "output": [ { "amount": 1.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000106" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/anthropic/claude-4-sonnet", "name": "replicate/anthropic/claude-4-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/deepseek-ai/deepseek-v3", "name": "replicate/deepseek-ai/deepseek-v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.4500000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000145" } ], "output": [ { "amount": 1.4500000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000145" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 65536, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/anthropic/claude-3.7-sonnet", "name": "replicate/anthropic/claude-3.7-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/anthropic/claude-3.5-haiku", "name": "replicate/anthropic/claude-3.5-haiku", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/anthropic/claude-3.5-sonnet", "name": "replicate/anthropic/claude-3.5-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "output": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/google/gemini-3-pro", "name": "replicate/google/gemini-3-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/anthropic/claude-4.5-sonnet", "name": "replicate/anthropic/claude-4.5-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/gpt-4.1", "name": "replicate/openai/gpt-4.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/gpt-4.1-nano", "name": "replicate/openai/gpt-4.1-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/gpt-4.1-mini", "name": "replicate/openai/gpt-4.1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/gpt-5-nano", "name": "replicate/openai/gpt-5-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/gpt-5-mini", "name": "replicate/openai/gpt-5-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/google/gemini-2.5-flash", "name": "replicate/google/gemini-2.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/openai/gpt-oss-120b", "name": "replicate/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/deepseek-ai/deepseek-v3.1", "name": "replicate/deepseek-ai/deepseek-v3.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6719999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.72e-7" } ], "output": [ { "amount": 2.016, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/xai/grok-4", "name": "replicate/xai/grok-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 7.199999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000072" } ], "output": [ { "amount": 36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "replicate/deepseek-ai/deepseek-r1", "name": "replicate/deepseek-ai/deepseek-r1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "replicate", "maxInputTokens": 65536, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "rerank-english-v2.0", "name": "rerank-english-v2.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "cohere", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "rerank-english-v3.0", "name": "rerank-english-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "cohere", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "rerank-multilingual-v2.0", "name": "rerank-multilingual-v2.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "cohere", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "rerank-multilingual-v3.0", "name": "rerank-multilingual-v3.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "cohere", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "rerank-v3.5", "name": "rerank-v3.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "cohere", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nvidia_nim/nvidia/nv-rerankqa-mistral-4b-v3", "name": "nvidia_nim/nvidia/nv-rerankqa-mistral-4b-v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "nvidia_nim", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nvidia_nim/nvidia/llama-3_2-nv-rerankqa-1b-v2", "name": "nvidia_nim/nvidia/llama-3_2-nv-rerankqa-1b-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "nvidia_nim", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "nvidia_nim/ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2", "name": "nvidia_nim/ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "nvidia_nim", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sagemaker/meta-textgeneration-llama-2-13b", "name": "sagemaker/meta-textgeneration-llama-2-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "sagemaker", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sagemaker/meta-textgeneration-llama-2-13b-f", "name": "sagemaker/meta-textgeneration-llama-2-13b-f", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sagemaker", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sagemaker/meta-textgeneration-llama-2-70b", "name": "sagemaker/meta-textgeneration-llama-2-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "sagemaker", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sagemaker/meta-textgeneration-llama-2-70b-b-f", "name": "sagemaker/meta-textgeneration-llama-2-70b-b-f", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sagemaker", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sagemaker/meta-textgeneration-llama-2-7b", "name": "sagemaker/meta-textgeneration-llama-2-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "sagemaker", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sagemaker/meta-textgeneration-llama-2-7b-f", "name": "sagemaker/meta-textgeneration-llama-2-7b-f", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sagemaker", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/MiniMax-M2.7", "name": "sambanova/MiniMax-M2.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 196608, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/DeepSeek-R1", "name": "sambanova/DeepSeek-R1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000007" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/DeepSeek-R1-Distill-Llama-70B", "name": "sambanova/DeepSeek-R1-Distill-Llama-70B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "output": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2026-03-20", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/DeepSeek-V3-0324", "name": "sambanova/DeepSeek-V3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/Llama-4-Maverick-17B-128E-Instruct", "name": "sambanova/Llama-4-Maverick-17B-128E-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.63, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.3e-7" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/Llama-4-Scout-17B-16E-Instruct", "name": "sambanova/Llama-4-Scout-17B-16E-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2025-06-19", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/Meta-Llama-3.1-405B-Instruct", "name": "sambanova/Meta-Llama-3.1-405B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2025-06-25", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/Meta-Llama-3.1-8B-Instruct", "name": "sambanova/Meta-Llama-3.1-8B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2026-04-14", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/Meta-Llama-3.2-1B-Instruct", "name": "sambanova/Meta-Llama-3.2-1B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2025-06-25", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/Meta-Llama-3.2-3B-Instruct", "name": "sambanova/Meta-Llama-3.2-3B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2025-06-25", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/Meta-Llama-3.3-70B-Instruct", "name": "sambanova/Meta-Llama-3.3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/Meta-Llama-Guard-3-8B", "name": "sambanova/Meta-Llama-Guard-3-8B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2025-06-25", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/QwQ-32B", "name": "sambanova/QwQ-32B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": "2025-06-25", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/Qwen2-Audio-7B-Instruct", "name": "sambanova/Qwen2-Audio-7B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 100, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2025-06-19", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/Qwen3-32B", "name": "sambanova/Qwen3-32B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-04-06", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/DeepSeek-V3.1", "name": "sambanova/DeepSeek-V3.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/gpt-oss-120b", "name": "sambanova/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.59, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/DeepSeek-V3.2", "name": "sambanova/DeepSeek-V3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sambanova/gemma-4-31B-it", "name": "sambanova/gemma-4-31B-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "output": [ { "amount": 1.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000115" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sambanova", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "snowflake/claude-3-5-sonnet", "name": "snowflake/claude-3-5-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 200000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/deepseek-r1", "name": "snowflake/deepseek-r1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/llama3.1-405b", "name": "snowflake/llama3.1-405b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/llama3.1-70b", "name": "snowflake/llama3.1-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/llama3.1-8b", "name": "snowflake/llama3.1-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/llama3.3-70b", "name": "snowflake/llama3.3-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/mistral-large2", "name": "snowflake/mistral-large2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/snowflake-llama-3.3-70b", "name": "snowflake/snowflake-llama-3.3-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "stability/sd3", "name": "stability/sd3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.065, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.065" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/sd3-large", "name": "stability/sd3-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.065, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.065" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/sd3-large-turbo", "name": "stability/sd3-large-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/sd3-medium", "name": "stability/sd3-medium", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.035, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.035" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/sd3.5-large", "name": "stability/sd3.5-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.065, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.065" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/sd3.5-large-turbo", "name": "stability/sd3.5-large-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/sd3.5-medium", "name": "stability/sd3.5-medium", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.035, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.035" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/stable-image-ultra", "name": "stability/stable-image-ultra", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.08, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.08" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/inpaint", "name": "stability/inpaint", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/outpaint", "name": "stability/outpaint", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.004, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.004" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/erase", "name": "stability/erase", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/search-and-replace", "name": "stability/search-and-replace", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/search-and-recolor", "name": "stability/search-and-recolor", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/remove-background", "name": "stability/remove-background", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/replace-background-and-relight", "name": "stability/replace-background-and-relight", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.008, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.008" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/sketch", "name": "stability/sketch", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/structure", "name": "stability/structure", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/style", "name": "stability/style", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.005" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/style-transfer", "name": "stability/style-transfer", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.008, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.008" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/fast", "name": "stability/fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.002, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.002" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/conservative", "name": "stability/conservative", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/creative", "name": "stability/creative", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.06, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.06" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability/stable-image-core", "name": "stability/stable-image-core", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.03, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.03" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "stability", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.sd3-5-large-v1:0", "name": "stability.sd3-5-large-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.08, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.08" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.sd3-large-v1:0", "name": "stability.sd3-large-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.08, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.08" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-core-v1:0", "name": "stability.stable-image-core-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-conservative-upscale-v1:0", "name": "stability.stable-conservative-upscale-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.4, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.4" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-creative-upscale-v1:0", "name": "stability.stable-creative-upscale-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.6, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.6" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-fast-upscale-v1:0", "name": "stability.stable-fast-upscale-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.03, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.03" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-outpaint-v1:0", "name": "stability.stable-outpaint-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.06, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.06" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-control-sketch-v1:0", "name": "stability.stable-image-control-sketch-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.07, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.07" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-control-structure-v1:0", "name": "stability.stable-image-control-structure-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.07, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.07" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-erase-object-v1:0", "name": "stability.stable-image-erase-object-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.07, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.07" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-inpaint-v1:0", "name": "stability.stable-image-inpaint-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.07, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.07" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-remove-background-v1:0", "name": "stability.stable-image-remove-background-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.07, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.07" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-search-recolor-v1:0", "name": "stability.stable-image-search-recolor-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.07, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.07" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-search-replace-v1:0", "name": "stability.stable-image-search-replace-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.07, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.07" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-style-guide-v1:0", "name": "stability.stable-image-style-guide-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.07, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.07" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-style-transfer-v1:0", "name": "stability.stable-style-transfer-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.08, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.08" } ] } ], "metadata": { "source": "litellm", "mode": "image_edit", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-core-v1:1", "name": "stability.stable-image-core-v1:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-ultra-v1:0", "name": "stability.stable-image-ultra-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.14, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.14" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "stability.stable-image-ultra-v1:1", "name": "stability.stable-image-ultra-v1:1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.14, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.14" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "bedrock", "maxInputTokens": 77, "maxOutputTokens": null, "maxTokens": 77, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "standard/1024-x-1024/dall-e-3", "name": "standard/1024-x-1024/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 3.81469e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "3.81469e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "standard/1024-x-1792/dall-e-3", "name": "standard/1024-x-1792/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.359e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "4.359e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "standard/1792-x-1024/dall-e-3", "name": "standard/1792-x-1024/dall-e-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 4.359e-8, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "4.359e-8" }, { "amount": 0, "currency": "USD", "units": 1, "pricingType": "pixel", "modality": "image", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "linkup/search", "name": "linkup/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00587, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.00587" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "linkup", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "linkup/search-deep", "name": "linkup/search-deep", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.05867, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.05867" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "linkup", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "tavily/search", "name": "tavily/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.008, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.008" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "tavily", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "tavily/search-advanced", "name": "tavily/search-advanced", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.016, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.016" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "tavily", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "you_com/search", "name": "you_com/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "you_com", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-completion-codestral/codestral-2405", "name": "text-completion-codestral/codestral-2405", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "text-completion-codestral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-completion-codestral/codestral-latest", "name": "text-completion-codestral/codestral-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "text-completion-codestral", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-embedding-004", "name": "text-embedding-004", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 2.5e-8, "currency": "USD", "units": 1, "pricingType": "character", "raw": "2.5e-8" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai-embedding-models", "maxInputTokens": 2048, "maxOutputTokens": null, "maxTokens": 2048, "deprecationDate": "2026-01-14", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-embedding-005", "name": "text-embedding-005", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 2.5e-8, "currency": "USD", "units": 1, "pricingType": "character", "raw": "2.5e-8" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai-embedding-models", "maxInputTokens": 2048, "maxOutputTokens": null, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-embedding-3-large", "name": "text-embedding-3-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.3e-7" }, { "amount": 0.065, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "6.5e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" }, { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "openai", "maxInputTokens": 8191, "maxOutputTokens": null, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-embedding-3-small", "name": "text-embedding-3-small", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-8" }, { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" }, { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "openai", "maxInputTokens": 8191, "maxOutputTokens": null, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-embedding-ada-002", "name": "text-embedding-ada-002", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "openai", "maxInputTokens": 8191, "maxOutputTokens": null, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-embedding-ada-002-v2", "name": "text-embedding-ada-002-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" }, { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "5e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" }, { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "openai", "maxInputTokens": 8191, "maxOutputTokens": null, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-embedding-large-exp-03-07", "name": "text-embedding-large-exp-03-07", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 2.5e-8, "currency": "USD", "units": 1, "pricingType": "character", "raw": "2.5e-8" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai-embedding-models", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-embedding-preview-0409", "name": "text-embedding-preview-0409", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.0062499999999999995, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "6.25e-9" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai-embedding-models", "maxInputTokens": 3072, "maxOutputTokens": null, "maxTokens": 3072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-moderation-007", "name": "text-moderation-007", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "moderation", "servingProvider": "openai", "maxInputTokens": 32768, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-moderation-latest", "name": "text-moderation-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "moderation", "servingProvider": "openai", "maxInputTokens": 32768, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-moderation-stable", "name": "text-moderation-stable", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "moderation", "servingProvider": "openai", "maxInputTokens": 32768, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-multilingual-embedding-002", "name": "text-multilingual-embedding-002", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 2.5e-8, "currency": "USD", "units": 1, "pricingType": "character", "raw": "2.5e-8" } ] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vertex_ai-embedding-models", "maxInputTokens": 2048, "maxOutputTokens": null, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-unicorn", "name": "text-unicorn", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "vertex_ai-text-models", "maxInputTokens": 8192, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "text-unicorn@001", "name": "text-unicorn@001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000028" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "completion", "servingProvider": "vertex_ai-text-models", "maxInputTokens": 8192, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together-ai-21.1b-41b", "name": "together-ai-21.1b-41b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together-ai-4.1b-8b", "name": "together-ai-4.1b-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together-ai-41.1b-80b", "name": "together-ai-41.1b-80b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together-ai-8.1b-21b", "name": "together-ai-8.1b-21b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": 1000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together-ai-81.1b-110b", "name": "together-ai-81.1b-110b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together-ai-embedding-151m-to-350m", "name": "together-ai-embedding-151m-to-350m", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.016, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.6e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together-ai-embedding-up-to-150m", "name": "together-ai-embedding-up-to-150m", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.008, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "8e-9" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/baai/bge-base-en-v1.5", "name": "together_ai/baai/bge-base-en-v1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.008, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "8e-9" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "together_ai", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/BAAI/bge-base-en-v1.5", "name": "together_ai/BAAI/bge-base-en-v1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.008, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "8e-9" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "together_ai", "maxInputTokens": 512, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together-ai-up-to-4b", "name": "together-ai-up-to-4b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/Qwen/Qwen3-235B-A22B-Instruct-2507-tput", "name": "together_ai/Qwen/Qwen3-235B-A22B-Instruct-2507-tput", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 262000, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "together_ai/Qwen/Qwen3-235B-A22B-Thinking-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.65, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.5e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 256000, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/Qwen/Qwen3-235B-A22B-fp8-tput", "name": "together_ai/Qwen/Qwen3-235B-A22B-fp8-tput", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 40000, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8", "name": "together_ai/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 256000, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/deepseek-ai/DeepSeek-R1", "name": "together_ai/deepseek-ai/DeepSeek-R1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000007" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 128000, "maxOutputTokens": 20480, "maxTokens": 20480, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/deepseek-ai/DeepSeek-R1-0528-tput", "name": "together_ai/deepseek-ai/DeepSeek-R1-0528-tput", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "output": [ { "amount": 2.1900000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000219" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/deepseek-ai/DeepSeek-V3", "name": "together_ai/deepseek-ai/DeepSeek-V3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 65536, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/deepseek-ai/DeepSeek-V3.1", "name": "together_ai/deepseek-ai/DeepSeek-V3.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000017" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/meta-llama/Llama-3.3-70B-Instruct-Turbo", "name": "together_ai/meta-llama/Llama-3.3-70B-Instruct-Turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/meta-llama/Llama-3.3-70B-Instruct-Turbo-Free", "name": "together_ai/meta-llama/Llama-3.3-70B-Instruct-Turbo-Free", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", "name": "together_ai/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/meta-llama/Llama-4-Scout-17B-16E-Instruct", "name": "together_ai/meta-llama/Llama-4-Scout-17B-16E-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "output": [ { "amount": 0.59, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo", "name": "together_ai/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "output": [ { "amount": 3.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000035" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo", "name": "together_ai/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo", "name": "together_ai/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "output": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/mistralai/Mixtral-8x7B-Instruct-v0.1", "name": "together_ai/mistralai/Mixtral-8x7B-Instruct-v0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/moonshotai/Kimi-K2-Instruct", "name": "together_ai/moonshotai/Kimi-K2-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/openai/gpt-oss-120b", "name": "together_ai/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/openai/gpt-oss-20b", "name": "together_ai/openai/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/zai-org/GLM-4.5-Air-FP8", "name": "together_ai/zai-org/GLM-4.5-Air-FP8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 128000, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/zai-org/GLM-4.6", "name": "together_ai/zai-org/GLM-4.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 200000, "maxOutputTokens": 200000, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/zai-org/GLM-4.7", "name": "together_ai/zai-org/GLM-4.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 200000, "maxOutputTokens": 200000, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/moonshotai/Kimi-K2.5", "name": "together_ai/moonshotai/Kimi-K2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 2.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000028" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/moonshotai/Kimi-K2-Instruct-0905", "name": "together_ai/moonshotai/Kimi-K2-Instruct-0905", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 262144, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/Qwen/Qwen3-Next-80B-A3B-Instruct", "name": "together_ai/Qwen/Qwen3-Next-80B-A3B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 262144, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/Qwen/Qwen3-Next-80B-A3B-Thinking", "name": "together_ai/Qwen/Qwen3-Next-80B-A3B-Thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 262144, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "together_ai/Qwen/Qwen3.5-397B-A17B", "name": "together_ai/Qwen/Qwen3.5-397B-A17B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "together_ai", "maxInputTokens": 262144, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "tts-1", "name": "tts-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000015, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.000015" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "tts-1-hd", "name": "tts-1-hd", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00003, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00003" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aws_polly/standard", "name": "aws_polly/standard", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000004, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.000004" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "aws_polly", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aws_polly/neural", "name": "aws_polly/neural", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000016, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.000016" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "aws_polly", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aws_polly/long-form", "name": "aws_polly/long-form", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.0001" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "aws_polly", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "aws_polly/generative", "name": "aws_polly/generative", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00003, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00003" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "aws_polly", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.amazon.nova-lite-v1:0", "name": "us.amazon.nova-lite-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.amazon.nova-micro-v1:0", "name": "us.amazon.nova-micro-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-8" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.amazon.nova-premier-v1:0", "name": "us.amazon.nova-premier-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": "2026-09-14", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": false, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.amazon.nova-pro-v1:0", "name": "us.amazon.nova-pro-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 300000, "maxOutputTokens": 10000, "maxTokens": 10000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-3-5-haiku-20241022-v1:0", "name": "us.anthropic.claude-3-5-haiku-20241022-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheWrite": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-haiku-4-5-20251001-v1:0", "name": "us.anthropic.claude-haiku-4-5-20251001-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" }, { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-3-5-sonnet-20240620-v1:0", "name": "us.anthropic.claude-3-5-sonnet-20240620-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-3-5-sonnet-20241022-v2:0", "name": "us.anthropic.claude-3-5-sonnet-20241022-v2:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-3-7-sonnet-20250219-v1:0", "name": "us.anthropic.claude-3-7-sonnet-20250219-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-3-haiku-20240307-v1:0", "name": "us.anthropic.claude-3-haiku-20240307-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [ { "amount": 0.3125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.125e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-09-10", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-3-opus-20240229-v1:0", "name": "us.anthropic.claude-3-opus-20240229-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-3-sonnet-20240229-v1:0", "name": "us.anthropic.claude-3-sonnet-20240229-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-07-30", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-opus-4-1-20250805-v1:0", "name": "us.anthropic.claude-opus-4-1-20250805-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": "2027-01-08", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "us.anthropic.claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.3000000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000033" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" }, { "amount": 24.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002475" } ], "cacheRead": [ { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" }, { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" } ], "cacheWrite": [ { "amount": 4.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004125" }, { "amount": 8.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000825" }, { "amount": 6.6000000000000005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000066" }, { "amount": 13.200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000132" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us-gov.anthropic.claude-sonnet-4-5-20250929-v1:0", "name": "us-gov.anthropic.claude-sonnet-4-5-20250929-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" }, { "amount": 7.199999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000072" } ], "output": [ { "amount": 18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000018" }, { "amount": 27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000027" } ], "cacheRead": [ { "amount": 0.36, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.6e-7" }, { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheWrite": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" }, { "amount": 9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000009" }, { "amount": 7.199999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000072" }, { "amount": 14.399999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000144" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "au.anthropic.claude-haiku-4-5-20251001-v1:0", "name": "au.anthropic.claude-haiku-4-5-20251001-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [ { "amount": 1.375, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001375" }, { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-opus-4-20250514-v1:0", "name": "us.anthropic.claude-opus-4-20250514-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-opus-4-5-20251101-v1:0", "name": "us.anthropic.claude-opus-4-5-20251101-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 27.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000275" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "global.anthropic.claude-opus-4-5-20251101-v1:0", "name": "global.anthropic.claude-opus-4-5-20251101-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "eu.anthropic.claude-opus-4-5-20251101-v1:0", "name": "eu.anthropic.claude-opus-4-5-20251101-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.anthropic.claude-sonnet-4-20250514-v1:0", "name": "us.anthropic.claude-sonnet-4-20250514-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": "2026-10-14", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "us.deepseek.r1-v1:0", "name": "us.deepseek.r1-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.deepseek.v3.2", "name": "us.deepseek.v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.62, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.2e-7" } ], "output": [ { "amount": 1.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000185" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "eu.deepseek.v3.2", "name": "eu.deepseek.v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.4e-7" } ], "output": [ { "amount": 2.2199999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000222" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.meta.llama3-1-405b-instruct-v1:0", "name": "us.meta.llama3-1-405b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000532" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.meta.llama3-1-70b-instruct-v1:0", "name": "us.meta.llama3-1-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.9900000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.9e-7" } ], "output": [ { "amount": 0.9900000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.meta.llama3-1-8b-instruct-v1:0", "name": "us.meta.llama3-1-8b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.meta.llama3-2-11b-instruct-v1:0", "name": "us.meta.llama3-2-11b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.meta.llama3-2-1b-instruct-v1:0", "name": "us.meta.llama3-2-1b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.meta.llama3-2-3b-instruct-v1:0", "name": "us.meta.llama3-2-3b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.meta.llama3-2-90b-instruct-v1:0", "name": "us.meta.llama3-2-90b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.meta.llama3-3-70b-instruct-v1:0", "name": "us.meta.llama3-3-70b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.meta.llama4-maverick-17b-instruct-v1:0", "name": "us.meta.llama4-maverick-17b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" }, { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0.9700000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.7e-7" }, { "amount": 0.48500000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.85e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.meta.llama4-scout-17b-instruct-v1:0", "name": "us.meta.llama4-scout-17b-instruct-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.7e-7" }, { "amount": 0.08499999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.5e-8" } ], "output": [ { "amount": 0.66, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6e-7" }, { "amount": 0.33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "us.mistral.pixtral-large-2502-v1:0", "name": "us.mistral.pixtral-large-2502-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "v0/v0-1.0-md", "name": "v0/v0-1.0-md", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "v0", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "v0/v0-1.5-lg", "name": "v0/v0-1.5-lg", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "v0", "maxInputTokens": 512000, "maxOutputTokens": 512000, "maxTokens": 512000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "v0/v0-1.5-md", "name": "v0/v0-1.5-md", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "v0", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "vercel_ai_gateway/alibaba/qwen-3-14b", "name": "vercel_ai_gateway/alibaba/qwen-3-14b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 40960, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/alibaba/qwen-3-235b", "name": "vercel_ai_gateway/alibaba/qwen-3-235b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 40960, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/alibaba/qwen-3-30b", "name": "vercel_ai_gateway/alibaba/qwen-3-30b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 40960, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/alibaba/qwen-3-32b", "name": "vercel_ai_gateway/alibaba/qwen-3-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 40960, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/alibaba/qwen3-coder", "name": "vercel_ai_gateway/alibaba/qwen3-coder", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 262144, "maxOutputTokens": 66536, "maxTokens": 66536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/amazon/nova-lite", "name": "vercel_ai_gateway/amazon/nova-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 300000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/amazon/nova-micro", "name": "vercel_ai_gateway/amazon/nova-micro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-8" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/amazon/nova-pro", "name": "vercel_ai_gateway/amazon/nova-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 300000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/amazon/titan-embed-text-v2", "name": "vercel_ai_gateway/amazon/titan-embed-text-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 0, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-3-haiku", "name": "vercel_ai_gateway/anthropic/claude-3-haiku", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-3-opus", "name": "vercel_ai_gateway/anthropic/claude-3-opus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-3.5-haiku", "name": "vercel_ai_gateway/anthropic/claude-3.5-haiku", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheWrite": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-3.5-sonnet", "name": "vercel_ai_gateway/anthropic/claude-3.5-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-3.7-sonnet", "name": "vercel_ai_gateway/anthropic/claude-3.7-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-4-opus", "name": "vercel_ai_gateway/anthropic/claude-4-opus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-4-sonnet", "name": "vercel_ai_gateway/anthropic/claude-4-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-3-5-sonnet", "name": "vercel_ai_gateway/anthropic/claude-3-5-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-3-5-sonnet-20241022", "name": "vercel_ai_gateway/anthropic/claude-3-5-sonnet-20241022", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-3-7-sonnet", "name": "vercel_ai_gateway/anthropic/claude-3-7-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-haiku-4.5", "name": "vercel_ai_gateway/anthropic/claude-haiku-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-opus-4", "name": "vercel_ai_gateway/anthropic/claude-opus-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-opus-4.1", "name": "vercel_ai_gateway/anthropic/claude-opus-4.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-opus-4.5", "name": "vercel_ai_gateway/anthropic/claude-opus-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-opus-4.6", "name": "vercel_ai_gateway/anthropic/claude-opus-4.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-sonnet-4", "name": "vercel_ai_gateway/anthropic/claude-sonnet-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/anthropic/claude-sonnet-4.5", "name": "vercel_ai_gateway/anthropic/claude-sonnet-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/cohere/command-a", "name": "vercel_ai_gateway/cohere/command-a", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 256000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/cohere/command-r", "name": "vercel_ai_gateway/cohere/command-r", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/cohere/command-r-plus", "name": "vercel_ai_gateway/cohere/command-r-plus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/cohere/embed-v4.0", "name": "vercel_ai_gateway/cohere/embed-v4.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 0, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/deepseek/deepseek-r1", "name": "vercel_ai_gateway/deepseek/deepseek-r1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "output": [ { "amount": 2.1900000000000004, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000219" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/deepseek/deepseek-r1-distill-llama-70b", "name": "vercel_ai_gateway/deepseek/deepseek-r1-distill-llama-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "output": [ { "amount": 0.9900000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/deepseek/deepseek-v3", "name": "vercel_ai_gateway/deepseek/deepseek-v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/google/gemini-2.0-flash", "name": "vercel_ai_gateway/google/gemini-2.0-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/google/gemini-2.0-flash-lite", "name": "vercel_ai_gateway/google/gemini-2.0-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/google/gemini-2.5-flash", "name": "vercel_ai_gateway/google/gemini-2.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 1000000, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/google/gemini-2.5-pro", "name": "vercel_ai_gateway/google/gemini-2.5-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/google/gemini-embedding-001", "name": "vercel_ai_gateway/google/gemini-embedding-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.5e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 0, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/google/gemma-2-9b", "name": "vercel_ai_gateway/google/gemma-2-9b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/google/text-embedding-005", "name": "vercel_ai_gateway/google/text-embedding-005", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2.5e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 0, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/google/text-multilingual-embedding-002", "name": "vercel_ai_gateway/google/text-multilingual-embedding-002", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2.5e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 0, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/inception/mercury-coder-small", "name": "vercel_ai_gateway/inception/mercury-coder-small", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 32000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/meta/llama-3-70b", "name": "vercel_ai_gateway/meta/llama-3-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.59, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.9e-7" } ], "output": [ { "amount": 0.7899999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/meta/llama-3-8b", "name": "vercel_ai_gateway/meta/llama-3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/meta/llama-3.1-70b", "name": "vercel_ai_gateway/meta/llama-3.1-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/meta/llama-3.1-8b", "name": "vercel_ai_gateway/meta/llama-3.1-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 131000, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/meta/llama-3.2-11b", "name": "vercel_ai_gateway/meta/llama-3.2-11b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "output": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/meta/llama-3.2-1b", "name": "vercel_ai_gateway/meta/llama-3.2-1b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/meta/llama-3.2-3b", "name": "vercel_ai_gateway/meta/llama-3.2-3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/meta/llama-3.2-90b", "name": "vercel_ai_gateway/meta/llama-3.2-90b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/meta/llama-3.3-70b", "name": "vercel_ai_gateway/meta/llama-3.3-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "output": [ { "amount": 0.72, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/meta/llama-4-maverick", "name": "vercel_ai_gateway/meta/llama-4-maverick", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/meta/llama-4-scout", "name": "vercel_ai_gateway/meta/llama-4-scout", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/codestral", "name": "vercel_ai_gateway/mistral/codestral", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 256000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/codestral-embed", "name": "vercel_ai_gateway/mistral/codestral-embed", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 0, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/devstral-small", "name": "vercel_ai_gateway/mistral/devstral-small", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/magistral-medium", "name": "vercel_ai_gateway/mistral/magistral-medium", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/magistral-small", "name": "vercel_ai_gateway/mistral/magistral-small", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/ministral-3b", "name": "vercel_ai_gateway/mistral/ministral-3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/ministral-8b", "name": "vercel_ai_gateway/mistral/ministral-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/mistral-embed", "name": "vercel_ai_gateway/mistral/mistral-embed", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 0, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/mistral-large", "name": "vercel_ai_gateway/mistral/mistral-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 32000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/mistral-saba-24b", "name": "vercel_ai_gateway/mistral/mistral-saba-24b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7899999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.9e-7" } ], "output": [ { "amount": 0.7899999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/mistral-small", "name": "vercel_ai_gateway/mistral/mistral-small", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 32000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/mixtral-8x22b-instruct", "name": "vercel_ai_gateway/mistral/mixtral-8x22b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 65536, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/pixtral-12b", "name": "vercel_ai_gateway/mistral/pixtral-12b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/mistral/pixtral-large", "name": "vercel_ai_gateway/mistral/pixtral-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/moonshotai/kimi-k2", "name": "vercel_ai_gateway/moonshotai/kimi-k2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 131072, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/morph/morph-v3-fast", "name": "vercel_ai_gateway/morph/morph-v3-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 32768, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/morph/morph-v3-large", "name": "vercel_ai_gateway/morph/morph-v3-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 1.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000019" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 32768, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/gpt-3.5-turbo", "name": "vercel_ai_gateway/openai/gpt-3.5-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 16385, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/gpt-3.5-turbo-instruct", "name": "vercel_ai_gateway/openai/gpt-3.5-turbo-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 8192, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/gpt-4-turbo", "name": "vercel_ai_gateway/openai/gpt-4-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/gpt-4.1", "name": "vercel_ai_gateway/openai/gpt-4.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/gpt-4.1-mini", "name": "vercel_ai_gateway/openai/gpt-4.1-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/gpt-4.1-nano", "name": "vercel_ai_gateway/openai/gpt-4.1-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 1047576, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/gpt-4o", "name": "vercel_ai_gateway/openai/gpt-4o", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/gpt-4o-mini", "name": "vercel_ai_gateway/openai/gpt-4o-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/o1", "name": "vercel_ai_gateway/openai/o1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00006" } ], "cacheRead": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/o3", "name": "vercel_ai_gateway/openai/o3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/o3-mini", "name": "vercel_ai_gateway/openai/o3-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/o4-mini", "name": "vercel_ai_gateway/openai/o4-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 100000, "maxTokens": 100000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/text-embedding-3-large", "name": "vercel_ai_gateway/openai/text-embedding-3-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.3e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 0, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/text-embedding-3-small", "name": "vercel_ai_gateway/openai/text-embedding-3-small", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 0, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/openai/text-embedding-ada-002", "name": "vercel_ai_gateway/openai/text-embedding-ada-002", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 0, "maxOutputTokens": 0, "maxTokens": 0, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/perplexity/sonar", "name": "vercel_ai_gateway/perplexity/sonar", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 127000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/perplexity/sonar-pro", "name": "vercel_ai_gateway/perplexity/sonar-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/perplexity/sonar-reasoning", "name": "vercel_ai_gateway/perplexity/sonar-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 127000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/perplexity/sonar-reasoning-pro", "name": "vercel_ai_gateway/perplexity/sonar-reasoning-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 127000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/vercel/v0-1.0-md", "name": "vercel_ai_gateway/vercel/v0-1.0-md", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/vercel/v0-1.5-md", "name": "vercel_ai_gateway/vercel/v0-1.5-md", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/xai/grok-2", "name": "vercel_ai_gateway/xai/grok-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 131072, "maxOutputTokens": 4000, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/xai/grok-2-vision", "name": "vercel_ai_gateway/xai/grok-2-vision", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/xai/grok-3", "name": "vercel_ai_gateway/xai/grok-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/xai/grok-3-fast", "name": "vercel_ai_gateway/xai/grok-3-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/xai/grok-3-mini", "name": "vercel_ai_gateway/xai/grok-3-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/xai/grok-3-mini-fast", "name": "vercel_ai_gateway/xai/grok-3-mini-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/xai/grok-4", "name": "vercel_ai_gateway/xai/grok-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/zai/glm-4.5", "name": "vercel_ai_gateway/zai/glm-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/zai/glm-4.5-air", "name": "vercel_ai_gateway/zai/glm-4.5-air", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 128000, "maxOutputTokens": 96000, "maxTokens": 96000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vercel_ai_gateway/zai/glm-4.6", "name": "vercel_ai_gateway/zai/glm-4.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vercel_ai_gateway", "maxInputTokens": 200000, "maxOutputTokens": 200000, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/chirp", "name": "vertex_ai/chirp", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00003, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00003" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "vertex_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/chirp_3", "name": "vertex_ai/chirp_3", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00026667, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00026667" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "vertex_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-3-5-haiku", "name": "vertex_ai/claude-3-5-haiku", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-3-5-haiku@20241022", "name": "vertex_ai/claude-3-5-haiku@20241022", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-haiku-4-5", "name": "vertex_ai/claude-haiku-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-haiku-4-5@20251001", "name": "vertex_ai/claude-haiku-4-5@20251001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-3-5-sonnet", "name": "vertex_ai/claude-3-5-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-3-5-sonnet@20240620", "name": "vertex_ai/claude-3-5-sonnet@20240620", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-3-7-sonnet@20250219", "name": "vertex_ai/claude-3-7-sonnet@20250219", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": "2026-05-11", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-3-haiku", "name": "vertex_ai/claude-3-haiku", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-3-haiku@20240307", "name": "vertex_ai/claude-3-haiku@20240307", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-3-opus", "name": "vertex_ai/claude-3-opus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-3-opus@20240229", "name": "vertex_ai/claude-3-opus@20240229", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-3-sonnet", "name": "vertex_ai/claude-3-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-3-sonnet@20240229", "name": "vertex_ai/claude-3-sonnet@20240229", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4", "name": "vertex_ai/claude-opus-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4-1", "name": "vertex_ai/claude-opus-4-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" }, { "amount": 37.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000375" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4-1@20250805", "name": "vertex_ai/claude-opus-4-1@20250805", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" }, { "amount": 37.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000375" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4-5", "name": "vertex_ai/claude-opus-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4-5@20251101", "name": "vertex_ai/claude-opus-4-5@20251101", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4-6", "name": "vertex_ai/claude-opus-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4-6@default", "name": "vertex_ai/claude-opus-4-6@default", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4-7", "name": "vertex_ai/claude-opus-4-7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4-7@default", "name": "vertex_ai/claude-opus-4-7@default", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-fable-5", "name": "vertex_ai/claude-fable-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-fable-5@default", "name": "vertex_ai/claude-fable-5@default", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-5", "name": "vertex_ai/claude-opus-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-5@default", "name": "vertex_ai/claude-opus-5@default", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4-8", "name": "vertex_ai/claude-opus-4-8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4-8@default", "name": "vertex_ai/claude-opus-4-8@default", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [ { "amount": 6.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000625" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-sonnet-4-5", "name": "vertex_ai/claude-sonnet-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-sonnet-5", "name": "vertex_ai/claude-sonnet-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-sonnet-4-6", "name": "vertex_ai/claude-sonnet-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-sonnet-4-5@20250929", "name": "vertex_ai/claude-sonnet-4-5@20250929", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-opus-4@20250514", "name": "vertex_ai/claude-opus-4@20250514", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "output": [ { "amount": 75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000075" } ], "cacheRead": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheWrite": [ { "amount": 18.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001875" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00003" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 200000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-sonnet-4", "name": "vertex_ai/claude-sonnet-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-sonnet-4@20250514", "name": "vertex_ai/claude-sonnet-4@20250514", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" }, { "amount": 22.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000225" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistralai/codestral-2@001", "name": "vertex_ai/mistralai/codestral-2@001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/codestral-2", "name": "vertex_ai/codestral-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/codestral-2@001", "name": "vertex_ai/codestral-2@001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistralai/codestral-2", "name": "vertex_ai/mistralai/codestral-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/codestral-2501", "name": "vertex_ai/codestral-2501", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/codestral@2405", "name": "vertex_ai/codestral@2405", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/codestral@latest", "name": "vertex_ai/codestral@latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/deepseek-ai/deepseek-v3.1-maas", "name": "vertex_ai/deepseek-ai/deepseek-v3.1-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-deepseek_models", "maxInputTokens": 163840, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/deepseek-ai/deepseek-v3.2-maas", "name": "vertex_ai/deepseek-ai/deepseek-v3.2-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.6e-7" }, { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "output": [ { "amount": 1.68, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000168" }, { "amount": 0.84, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-deepseek_models", "maxInputTokens": 163840, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/deepseek-ai/deepseek-r1-0528-maas", "name": "vertex_ai/deepseek-ai/deepseek-r1-0528-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000135" } ], "output": [ { "amount": 5.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000054" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-deepseek_models", "maxInputTokens": 65336, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/gemini-2.5-flash-image", "name": "vertex_ai/gemini-2.5-flash-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" }, { "amount": 30, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00003" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "3e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.0000025" }, { "amount": 0.039, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.039" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": false, "responseSchema": true, "systemMessages": true, "webSearch": false } } }, { "id": "vertex_ai/gemini-3-pro-image", "name": "vertex_ai/gemini-3-pro-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000006" }, { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0011, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0011" }, { "amount": 0.134, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.134" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": false, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/gemini-3-pro-image-preview", "name": "vertex_ai/gemini-3-pro-image-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000006" }, { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0011, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0011" }, { "amount": 0.134, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.134" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": false, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/gemini-3.1-flash-image", "name": "vertex_ai/gemini-3.1-flash-image", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "5e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00056, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00056" }, { "amount": 0.0672, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0672" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": false, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/gemini-3.1-flash-image-preview", "name": "vertex_ai/gemini-3.1-flash-image-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "5e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000003" }, { "amount": 60, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00006" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00056, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.00056" }, { "amount": 0.0672, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0672" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": false, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/gemini-3.1-flash-lite-preview", "name": "vertex_ai/gemini-3.1-flash-lite-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "vertex_ai/gemini-3.1-flash-lite", "name": "vertex_ai/gemini-3.1-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" }, { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" }, { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 2.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000027" } ], "cacheRead": [ { "amount": 0.024999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-8" }, { "amount": 0.045, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-8" } ], "cacheWrite": [], "other": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "vertex_ai/gemini-3.5-flash-lite", "name": "vertex_ai/gemini-3.5-flash-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.54, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.4e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" }, { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 1048576, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "vertex_ai/deep-research-pro-preview-12-2025", "name": "vertex_ai/deep-research-pro-preview-12-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000002" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000012" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000006" }, { "amount": 120, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0011, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.0011" }, { "amount": 0.134, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.134" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-language-models", "maxInputTokens": 65536, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/imagegeneration@006", "name": "vertex_ai/imagegeneration@006", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.02" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-image-models", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/imagen-3.0-fast-generate-001", "name": "vertex_ai/imagen-3.0-fast-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.02" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-image-models", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/imagen-3.0-generate-001", "name": "vertex_ai/imagen-3.0-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-image-models", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/imagen-3.0-generate-002", "name": "vertex_ai/imagen-3.0-generate-002", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-image-models", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2025-11-10", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/imagen-3.0-capability-001", "name": "vertex_ai/imagen-3.0-capability-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-image-models", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/imagen-4.0-fast-generate-001", "name": "vertex_ai/imagen-4.0-fast-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.02" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-image-models", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/imagen-4.0-generate-001", "name": "vertex_ai/imagen-4.0-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.04, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.04" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-image-models", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/imagen-4.0-ultra-generate-001", "name": "vertex_ai/imagen-4.0-ultra-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.06, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.06" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "vertex_ai-image-models", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/jamba-1.5", "name": "vertex_ai/jamba-1.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-ai21_models", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/jamba-1.5-large", "name": "vertex_ai/jamba-1.5-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-ai21_models", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/jamba-1.5-large@001", "name": "vertex_ai/jamba-1.5-large@001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-ai21_models", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/jamba-1.5-mini", "name": "vertex_ai/jamba-1.5-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-ai21_models", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/jamba-1.5-mini@001", "name": "vertex_ai/jamba-1.5-mini@001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-ai21_models", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/meta/llama-3.1-405b-instruct-maas", "name": "vertex_ai/meta/llama-3.1-405b-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000016" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-llama_models", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "vertex_ai/meta/llama-3.1-70b-instruct-maas", "name": "vertex_ai/meta/llama-3.1-70b-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-llama_models", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "vertex_ai/meta/llama-3.1-8b-instruct-maas", "name": "vertex_ai/meta/llama-3.1-8b-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-llama_models", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "vertex_ai/meta/llama-3.2-90b-vision-instruct-maas", "name": "vertex_ai/meta/llama-3.2-90b-vision-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-llama_models", "maxInputTokens": 128000, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas", "name": "vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 1.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000115" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-llama_models", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas", "name": "vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 1.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000115" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-llama_models", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/meta/llama-4-scout-17b-128e-instruct-maas", "name": "vertex_ai/meta/llama-4-scout-17b-128e-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-llama_models", "maxInputTokens": 10000000, "maxOutputTokens": 10000000, "maxTokens": 10000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas", "name": "vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-llama_models", "maxInputTokens": 10000000, "maxOutputTokens": 10000000, "maxTokens": 10000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/meta/llama3-405b-instruct-maas", "name": "vertex_ai/meta/llama3-405b-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-llama_models", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/meta/llama3-70b-instruct-maas", "name": "vertex_ai/meta/llama3-70b-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-llama_models", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/meta/llama3-8b-instruct-maas", "name": "vertex_ai/meta/llama3-8b-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-llama_models", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/minimaxai/minimax-m2-maas", "name": "vertex_ai/minimaxai/minimax-m2-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-minimax_models", "maxInputTokens": 196608, "maxOutputTokens": 196608, "maxTokens": 196608, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/moonshotai/kimi-k2-thinking-maas", "name": "vertex_ai/moonshotai/kimi-k2-thinking-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-moonshot_models", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "vertex_ai/zai-org/glm-4.7-maas", "name": "vertex_ai/zai-org/glm-4.7-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-zai_models", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/zai-org/glm-5-maas", "name": "vertex_ai/zai-org/glm-5-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-zai_models", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistral-medium-3", "name": "vertex_ai/mistral-medium-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistral-medium-3@001", "name": "vertex_ai/mistral-medium-3@001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistralai/mistral-medium-3", "name": "vertex_ai/mistralai/mistral-medium-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistralai/mistral-medium-3@001", "name": "vertex_ai/mistralai/mistral-medium-3@001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistral-large-2411", "name": "vertex_ai/mistral-large-2411", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistral-large@2407", "name": "vertex_ai/mistral-large@2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistral-large@2411-001", "name": "vertex_ai/mistral-large@2411-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistral-large@latest", "name": "vertex_ai/mistral-large@latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistral-nemo@2407", "name": "vertex_ai/mistral-nemo@2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistral-nemo@latest", "name": "vertex_ai/mistral-nemo@latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistral-small-2503", "name": "vertex_ai/mistral-small-2503", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistral-small-2503@001", "name": "vertex_ai/mistral-small-2503@001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-mistral_models", "maxInputTokens": 32000, "maxOutputTokens": 8191, "maxTokens": 8191, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/mistral-ocr-2505", "name": "vertex_ai/mistral-ocr-2505", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0005, "currency": "USD", "units": 1, "pricingType": "page", "raw": "0.0005" } ] } ], "metadata": { "source": "litellm", "mode": "ocr", "servingProvider": "vertex_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/deepseek-ai/deepseek-ocr-maas", "name": "vertex_ai/deepseek-ai/deepseek-ocr-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0003, "currency": "USD", "units": 1, "pricingType": "page", "raw": "0.0003" } ] } ], "metadata": { "source": "litellm", "mode": "ocr", "servingProvider": "vertex_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/google/gemma-4-26b-a4b-it-maas", "name": "vertex_ai/google/gemma-4-26b-a4b-it-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-openai_models", "maxInputTokens": 256000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/openai/gpt-oss-120b-maas", "name": "vertex_ai/openai/gpt-oss-120b-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-openai_models", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/openai/gpt-oss-20b-maas", "name": "vertex_ai/openai/gpt-oss-20b-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-openai_models", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/xai/grok-4.1-fast-non-reasoning", "name": "vertex_ai/xai/grok-4.1-fast-non-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai", "maxInputTokens": 2000000, "maxOutputTokens": 2000000, "maxTokens": 2000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "vertex_ai/xai/grok-4.1-fast-reasoning", "name": "vertex_ai/xai/grok-4.1-fast-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai", "maxInputTokens": 2000000, "maxOutputTokens": 2000000, "maxTokens": 2000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "vertex_ai/xai/grok-4.20-non-reasoning", "name": "vertex_ai/xai/grok-4.20-non-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai", "maxInputTokens": 2000000, "maxOutputTokens": 2000000, "maxTokens": 2000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "vertex_ai/xai/grok-4.20-reasoning", "name": "vertex_ai/xai/grok-4.20-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai", "maxInputTokens": 2000000, "maxOutputTokens": 2000000, "maxTokens": 2000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "vertex_ai/qwen/qwen3-235b-a22b-instruct-2507-maas", "name": "vertex_ai/qwen/qwen3-235b-a22b-instruct-2507-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-qwen_models", "maxInputTokens": 262144, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/qwen/qwen3-coder-480b-a35b-instruct-maas", "name": "vertex_ai/qwen/qwen3-coder-480b-a35b-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-qwen_models", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/qwen/qwen3-next-80b-a3b-instruct-maas", "name": "vertex_ai/qwen/qwen3-next-80b-a3b-instruct-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-qwen_models", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/qwen/qwen3-next-80b-a3b-thinking-maas", "name": "vertex_ai/qwen/qwen3-next-80b-a3b-thinking-maas", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-qwen_models", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/veo-2.0-generate-001", "name": "vertex_ai/veo-2.0-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.35, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.35" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "vertex_ai-video-models", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/veo-3.0-fast-generate-001", "name": "vertex_ai/veo-3.0-fast-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.15, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.15" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "vertex_ai-video-models", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/veo-3.0-generate-001", "name": "vertex_ai/veo-3.0-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.4, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.4" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "vertex_ai-video-models", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/veo-3.1-generate-preview", "name": "vertex_ai/veo-3.1-generate-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.4, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.4" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "vertex_ai-video-models", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/veo-3.1-fast-generate-preview", "name": "vertex_ai/veo-3.1-fast-generate-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.15, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.15" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "vertex_ai-video-models", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/veo-3.1-generate-001", "name": "vertex_ai/veo-3.1-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.4, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.4" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "vertex_ai-video-models", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/veo-3.1-fast-generate-001", "name": "vertex_ai/veo-3.1-fast-generate-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.15, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.15" } ] } ], "metadata": { "source": "litellm", "mode": "video_generation", "servingProvider": "vertex_ai-video-models", "maxInputTokens": 1024, "maxOutputTokens": null, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/rerank-2", "name": "voyage/rerank-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "5e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "voyage", "maxInputTokens": 16000, "maxOutputTokens": 16000, "maxTokens": 16000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/rerank-2-lite", "name": "voyage/rerank-2-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "2e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "voyage", "maxInputTokens": 8000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/rerank-2.5", "name": "voyage/rerank-2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "5e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "voyage", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/rerank-2.5-lite", "name": "voyage/rerank-2.5-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "2e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "voyage", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-2", "name": "voyage/voyage-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 4000, "maxOutputTokens": null, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-3", "name": "voyage/voyage-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "6e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 32000, "maxOutputTokens": null, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-3-large", "name": "voyage/voyage-3-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.8e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 32000, "maxOutputTokens": null, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-3-lite", "name": "voyage/voyage-3-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 32000, "maxOutputTokens": null, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-3.5", "name": "voyage/voyage-3.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "6e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 32000, "maxOutputTokens": null, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-3.5-lite", "name": "voyage/voyage-3.5-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 32000, "maxOutputTokens": null, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-code-2", "name": "voyage/voyage-code-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 16000, "maxOutputTokens": null, "maxTokens": 16000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-code-3", "name": "voyage/voyage-code-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.8e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 32000, "maxOutputTokens": null, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-context-3", "name": "voyage/voyage-context-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.8e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 120000, "maxOutputTokens": null, "maxTokens": 120000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-finance-2", "name": "voyage/voyage-finance-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 32000, "maxOutputTokens": null, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-large-2", "name": "voyage/voyage-large-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 16000, "maxOutputTokens": null, "maxTokens": 16000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-law-2", "name": "voyage/voyage-law-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 16000, "maxOutputTokens": null, "maxTokens": 16000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-lite-01", "name": "voyage/voyage-lite-01", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 4096, "maxOutputTokens": null, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-lite-02-instruct", "name": "voyage/voyage-lite-02-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 4000, "maxOutputTokens": null, "maxTokens": 4000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "voyage/voyage-multimodal-3", "name": "voyage/voyage-multimodal-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1.2e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "voyage", "maxInputTokens": 32000, "maxOutputTokens": null, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/openai/gpt-oss-120b", "name": "wandb/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 15000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.015" } ], "output": [ { "amount": 60000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.06" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/openai/gpt-oss-20b", "name": "wandb/openai/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.005" } ], "output": [ { "amount": 20000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.02" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/zai-org/GLM-4.5", "name": "wandb/zai-org/GLM-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 55000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.055" } ], "output": [ { "amount": 200000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.2" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/Qwen/Qwen3-235B-A22B-Instruct-2507", "name": "wandb/Qwen/Qwen3-235B-A22B-Instruct-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.01" } ], "output": [ { "amount": 10000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.01" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/Qwen/Qwen3-Coder-480B-A35B-Instruct", "name": "wandb/Qwen/Qwen3-Coder-480B-A35B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 100000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.1" } ], "output": [ { "amount": 150000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.15" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "wandb/Qwen/Qwen3-235B-A22B-Thinking-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.01" } ], "output": [ { "amount": 10000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.01" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/moonshotai/Kimi-K2-Instruct", "name": "wandb/moonshotai/Kimi-K2-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/moonshotai/Kimi-K2.5", "name": "wandb/moonshotai/Kimi-K2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/MiniMaxAI/MiniMax-M2.5", "name": "wandb/MiniMaxAI/MiniMax-M2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 197000, "maxOutputTokens": 197000, "maxTokens": 197000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/meta-llama/Llama-3.1-8B-Instruct", "name": "wandb/meta-llama/Llama-3.1-8B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 22000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.022" } ], "output": [ { "amount": 22000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/deepseek-ai/DeepSeek-V3.1", "name": "wandb/deepseek-ai/DeepSeek-V3.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 55000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.055" } ], "output": [ { "amount": 165000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.165" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/deepseek-ai/DeepSeek-R1-0528", "name": "wandb/deepseek-ai/DeepSeek-R1-0528", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 135000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.135" } ], "output": [ { "amount": 540000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.54" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 161000, "maxOutputTokens": 161000, "maxTokens": 161000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/deepseek-ai/DeepSeek-V3-0324", "name": "wandb/deepseek-ai/DeepSeek-V3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 114000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.114" } ], "output": [ { "amount": 275000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.275" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 161000, "maxOutputTokens": 161000, "maxTokens": 161000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/meta-llama/Llama-3.3-70B-Instruct", "name": "wandb/meta-llama/Llama-3.3-70B-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 71000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.071" } ], "output": [ { "amount": 71000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.071" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/meta-llama/Llama-4-Scout-17B-16E-Instruct", "name": "wandb/meta-llama/Llama-4-Scout-17B-16E-Instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 17000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.017" } ], "output": [ { "amount": 66000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.066" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 64000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "wandb/microsoft/Phi-4-mini-instruct", "name": "wandb/microsoft/Phi-4-mini-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 8000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.008" } ], "output": [ { "amount": 35000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.035" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "wandb", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/ibm/granite-3-8b-instruct", "name": "watsonx/ibm/granite-3-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 1024, "maxTokens": 1024, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": false, "audioInput": false, "audioOutput": false, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "watsonx/mistralai/mistral-large", "name": "watsonx/mistralai/mistral-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 131072, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": false, "audioInput": false, "audioOutput": false, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "watsonx/bigscience/mt0-xxl-13b", "name": "watsonx/bigscience/mt0-xxl-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 500, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0005" } ], "output": [ { "amount": 2000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/core42/jais-13b-chat", "name": "watsonx/core42/jais-13b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 500, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0005" } ], "output": [ { "amount": 2000, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/google/flan-t5-xl-3b", "name": "watsonx/google/flan-t5-xl-3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/ibm/granite-13b-chat-v2", "name": "watsonx/ibm/granite-13b-chat-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/ibm/granite-13b-instruct-v2", "name": "watsonx/ibm/granite-13b-instruct-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/ibm/granite-3-3-8b-instruct", "name": "watsonx/ibm/granite-3-3-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/ibm/granite-4-h-small", "name": "watsonx/ibm/granite-4-h-small", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 20480, "maxOutputTokens": 20480, "maxTokens": 20480, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/ibm/granite-guardian-3-2-2b", "name": "watsonx/ibm/granite-guardian-3-2-2b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/ibm/granite-guardian-3-3-8b", "name": "watsonx/ibm/granite-guardian-3-3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/ibm/granite-ttm-1024-96-r2", "name": "watsonx/ibm/granite-ttm-1024-96-r2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "output": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 512, "maxOutputTokens": 512, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/ibm/granite-ttm-1536-96-r2", "name": "watsonx/ibm/granite-ttm-1536-96-r2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "output": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 512, "maxOutputTokens": 512, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/ibm/granite-ttm-512-96-r2", "name": "watsonx/ibm/granite-ttm-512-96-r2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "output": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 512, "maxOutputTokens": 512, "maxTokens": 512, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/ibm/granite-vision-3-2-2b", "name": "watsonx/ibm/granite-vision-3-2-2b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/meta-llama/llama-3-2-11b-vision-instruct", "name": "watsonx/meta-llama/llama-3-2-11b-vision-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/meta-llama/llama-3-2-1b-instruct", "name": "watsonx/meta-llama/llama-3-2-1b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/meta-llama/llama-3-2-3b-instruct", "name": "watsonx/meta-llama/llama-3-2-3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/meta-llama/llama-3-2-90b-vision-instruct", "name": "watsonx/meta-llama/llama-3-2-90b-vision-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/meta-llama/llama-3-3-70b-instruct", "name": "watsonx/meta-llama/llama-3-3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.71, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.1e-7" } ], "output": [ { "amount": 0.71, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/meta-llama/llama-4-maverick-17b", "name": "watsonx/meta-llama/llama-4-maverick-17b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/meta-llama/llama-guard-3-11b-vision", "name": "watsonx/meta-llama/llama-guard-3-11b-vision", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/mistralai/mistral-medium-2505", "name": "watsonx/mistralai/mistral-medium-2505", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/mistralai/mistral-small-2503", "name": "watsonx/mistralai/mistral-small-2503", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/mistralai/mistral-small-3-1-24b-instruct-2503", "name": "watsonx/mistralai/mistral-small-3-1-24b-instruct-2503", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/mistralai/pixtral-12b-2409", "name": "watsonx/mistralai/pixtral-12b-2409", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/openai/gpt-oss-120b", "name": "watsonx/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/sdaia/allam-1-13b-instruct", "name": "watsonx/sdaia/allam-1-13b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "watsonx", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": false, "parallelFunctionCalling": false, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "watsonx/whisper-large-v3-turbo", "name": "watsonx/whisper-large-v3-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" }, { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "watsonx", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "whisper-1", "name": "whisper-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" }, { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "xai/grok-2", "name": "xai/grok-2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-2-1212", "name": "xai/grok-2-1212", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-2-latest", "name": "xai/grok-2-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-2-vision", "name": "xai/grok-2-vision", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000002, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000002" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-2-vision-1212", "name": "xai/grok-2-vision-1212", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000002, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000002" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": "2026-02-28", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-2-vision-latest", "name": "xai/grok-2-vision-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000002, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000002" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-3", "name": "xai/grok-3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2026-05-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-3-beta", "name": "xai/grok-3-beta", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-3-fast-beta", "name": "xai/grok-3-fast-beta", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-3-fast-latest", "name": "xai/grok-3-fast-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-3-latest", "name": "xai/grok-3-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-3-mini", "name": "xai/grok-3-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2026-02-28", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-3-mini-beta", "name": "xai/grok-3-mini-beta", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": "2026-02-28", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-3-mini-fast", "name": "xai/grok-3-mini-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-3-mini-fast-beta", "name": "xai/grok-3-mini-fast-beta", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-3-mini-fast-latest", "name": "xai/grok-3-mini-fast-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-3-mini-latest", "name": "xai/grok-3-mini-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": false, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4", "name": "xai/grok-4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4-fast-reasoning", "name": "xai/grok-4-fast-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 2000000, "maxOutputTokens": 2000000, "maxTokens": 2000000, "deprecationDate": "2026-05-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4-fast-non-reasoning", "name": "xai/grok-4-fast-non-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 2000000, "maxOutputTokens": 2000000, "maxTokens": 2000000, "deprecationDate": "2026-05-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4-0709", "name": "xai/grok-4-0709", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": "2026-05-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4-latest", "name": "xai/grok-4-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4-1-fast", "name": "xai/grok-4-1-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 2000000, "maxOutputTokens": 2000000, "maxTokens": 2000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4-1-fast-reasoning", "name": "xai/grok-4-1-fast-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 2000000, "maxOutputTokens": 2000000, "maxTokens": 2000000, "deprecationDate": "2026-05-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4-1-fast-reasoning-latest", "name": "xai/grok-4-1-fast-reasoning-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 2000000, "maxOutputTokens": 2000000, "maxTokens": 2000000, "deprecationDate": "2026-05-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4-1-fast-non-reasoning", "name": "xai/grok-4-1-fast-non-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 2000000, "maxOutputTokens": 2000000, "maxTokens": 2000000, "deprecationDate": "2026-05-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4-1-fast-non-reasoning-latest", "name": "xai/grok-4-1-fast-non-reasoning-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 2000000, "maxOutputTokens": 2000000, "maxTokens": 2000000, "deprecationDate": "2026-05-15", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4.20-multi-agent-beta-0309", "name": "xai/grok-4.20-multi-agent-beta-0309", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4.20-beta-0309-reasoning", "name": "xai/grok-4.20-beta-0309-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4.20-0309-reasoning", "name": "xai/grok-4.20-0309-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4.20-beta-0309-non-reasoning", "name": "xai/grok-4.20-beta-0309-non-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4.3", "name": "xai/grok-4.3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4.3-latest", "name": "xai/grok-4.3-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4.5", "name": "xai/grok-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 500000, "maxOutputTokens": 500000, "maxTokens": 500000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4.5-latest", "name": "xai/grok-4.5-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 500000, "maxOutputTokens": 500000, "maxTokens": 500000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4.6", "name": "xai/grok-4.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000012" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 500000, "maxOutputTokens": 500000, "maxTokens": 500000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-beta", "name": "xai/grok-beta", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-code-fast", "name": "xai/grok-code-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "xai/grok-code-fast-1", "name": "xai/grok-code-fast-1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "xai/grok-code-fast-1-0825", "name": "xai/grok-code-fast-1-0825", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "xai/grok-vision-beta", "name": "xai/grok-vision-beta", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000005, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.000005" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "zai.glm-4.7", "name": "zai.glm-4.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "zai.glm-5", "name": "zai.glm-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "zai.glm-4.7-flash", "name": "zai.glm-4.7-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_converse", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "zai/glm-5", "name": "zai/glm-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-5.1", "name": "zai/glm-5.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.6e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-5-code", "name": "zai/glm-5-code", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-4.7", "name": "zai/glm-4.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-4.7-flash", "name": "zai/glm-4.7-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-4.6", "name": "zai/glm-4.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-4.5", "name": "zai/glm-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 128000, "maxOutputTokens": 32000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-4.5v", "name": "zai/glm-4.5v", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 128000, "maxOutputTokens": 32000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-4.5-x", "name": "zai/glm-4.5-x", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "output": [ { "amount": 8.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000089" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 128000, "maxOutputTokens": 32000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-4.5-air", "name": "zai/glm-4.5-air", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 128000, "maxOutputTokens": 32000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-4.5-airx", "name": "zai/glm-4.5-airx", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 128000, "maxOutputTokens": 32000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-4-32b-0414-128k", "name": "zai/glm-4-32b-0414-128k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 128000, "maxOutputTokens": 32000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "zai/glm-4.5-flash", "name": "zai/glm-4.5-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "zai", "maxInputTokens": 128000, "maxOutputTokens": 32000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/search_api", "name": "vertex_ai/search_api", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0015, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0.0015" } ] } ], "metadata": { "source": "litellm", "mode": "vector_store", "servingProvider": "vertex_ai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "openai/container", "name": "openai/container", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.03, "currency": "USD", "units": 1, "pricingType": "session", "raw": "0.03" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "runwayml/gen4_image", "name": "runwayml/gen4_image", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.05, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.05" }, { "amount": 0.05, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.05" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "runwayml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "runwayml/gen4_image_turbo", "name": "runwayml/gen4_image_turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.02, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.02" }, { "amount": 0.02, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.02" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "runwayml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "runwayml/eleven_multilingual_v2", "name": "runwayml/eleven_multilingual_v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 3e-7, "currency": "USD", "units": 1, "pricingType": "character", "raw": "3e-7" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "runwayml", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-coder-480b-a35b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen3-coder-480b-a35b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/flux-kontext-pro", "name": "fireworks_ai/accounts/fireworks/models/flux-kontext-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "4e-8" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "4e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/SSD-1B", "name": "fireworks_ai/accounts/fireworks/models/SSD-1B", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.00013, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.3e-10" } ], "output": [ { "amount": 0.00013, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.3e-10" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/chronos-hermes-13b-v2", "name": "fireworks_ai/accounts/fireworks/models/chronos-hermes-13b-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-13b", "name": "fireworks_ai/accounts/fireworks/models/code-llama-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-13b-instruct", "name": "fireworks_ai/accounts/fireworks/models/code-llama-13b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-13b-python", "name": "fireworks_ai/accounts/fireworks/models/code-llama-13b-python", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-34b", "name": "fireworks_ai/accounts/fireworks/models/code-llama-34b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-34b-instruct", "name": "fireworks_ai/accounts/fireworks/models/code-llama-34b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-34b-python", "name": "fireworks_ai/accounts/fireworks/models/code-llama-34b-python", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-70b", "name": "fireworks_ai/accounts/fireworks/models/code-llama-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-70b-instruct", "name": "fireworks_ai/accounts/fireworks/models/code-llama-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-70b-python", "name": "fireworks_ai/accounts/fireworks/models/code-llama-70b-python", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-7b", "name": "fireworks_ai/accounts/fireworks/models/code-llama-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-7b-instruct", "name": "fireworks_ai/accounts/fireworks/models/code-llama-7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-llama-7b-python", "name": "fireworks_ai/accounts/fireworks/models/code-llama-7b-python", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/code-qwen-1p5-7b", "name": "fireworks_ai/accounts/fireworks/models/code-qwen-1p5-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 65536, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/codegemma-2b", "name": "fireworks_ai/accounts/fireworks/models/codegemma-2b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/codegemma-7b", "name": "fireworks_ai/accounts/fireworks/models/codegemma-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/cogito-671b-v2-p1", "name": "fireworks_ai/accounts/fireworks/models/cogito-671b-v2-p1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-3b", "name": "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-70b", "name": "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-8b", "name": "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-qwen-14b", "name": "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-qwen-14b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-qwen-32b", "name": "fireworks_ai/accounts/fireworks/models/cogito-v1-preview-qwen-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/flux-kontext-max", "name": "fireworks_ai/accounts/fireworks/models/flux-kontext-max", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "8e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/dbrx-instruct", "name": "fireworks_ai/accounts/fireworks/models/dbrx-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-coder-1b-base", "name": "fireworks_ai/accounts/fireworks/models/deepseek-coder-1b-base", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-coder-33b-instruct", "name": "fireworks_ai/accounts/fireworks/models/deepseek-coder-33b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-base", "name": "fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-base", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-base-v1p5", "name": "fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-base-v1p5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-instruct-v1p5", "name": "fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-instruct-v1p5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-coder-v2-lite-base", "name": "fireworks_ai/accounts/fireworks/models/deepseek-coder-v2-lite-base", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-coder-v2-lite-instruct", "name": "fireworks_ai/accounts/fireworks/models/deepseek-coder-v2-lite-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-prover-v2", "name": "fireworks_ai/accounts/fireworks/models/deepseek-prover-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-r1-0528-distill-qwen3-8b", "name": "fireworks_ai/accounts/fireworks/models/deepseek-r1-0528-distill-qwen3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-llama-70b", "name": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-llama-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-llama-8b", "name": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-llama-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-14b", "name": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-14b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-1p5b", "name": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-1p5b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-32b", "name": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-7b", "name": "fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-v2-lite-chat", "name": "fireworks_ai/accounts/fireworks/models/deepseek-v2-lite-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/deepseek-v2p5", "name": "fireworks_ai/accounts/fireworks/models/deepseek-v2p5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/devstral-small-2505", "name": "fireworks_ai/accounts/fireworks/models/devstral-small-2505", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/dobby-mini-unhinged-plus-llama-3-1-8b", "name": "fireworks_ai/accounts/fireworks/models/dobby-mini-unhinged-plus-llama-3-1-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/dobby-unhinged-llama-3-3-70b-new", "name": "fireworks_ai/accounts/fireworks/models/dobby-unhinged-llama-3-3-70b-new", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/dolphin-2-9-2-qwen2-72b", "name": "fireworks_ai/accounts/fireworks/models/dolphin-2-9-2-qwen2-72b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/dolphin-2p6-mixtral-8x7b", "name": "fireworks_ai/accounts/fireworks/models/dolphin-2p6-mixtral-8x7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/ernie-4p5-21b-a3b-pt", "name": "fireworks_ai/accounts/fireworks/models/ernie-4p5-21b-a3b-pt", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/ernie-4p5-300b-a47b-pt", "name": "fireworks_ai/accounts/fireworks/models/ernie-4p5-300b-a47b-pt", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/fare-20b", "name": "fireworks_ai/accounts/fireworks/models/fare-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/firefunction-v1", "name": "fireworks_ai/accounts/fireworks/models/firefunction-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/firellava-13b", "name": "fireworks_ai/accounts/fireworks/models/firellava-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/firesearch-ocr-v6", "name": "fireworks_ai/accounts/fireworks/models/firesearch-ocr-v6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/fireworks-asr-large", "name": "fireworks_ai/accounts/fireworks/models/fireworks-asr-large", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/fireworks-asr-v2", "name": "fireworks_ai/accounts/fireworks/models/fireworks-asr-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/flux-1-dev", "name": "fireworks_ai/accounts/fireworks/models/flux-1-dev", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/flux-1-dev-controlnet-union", "name": "fireworks_ai/accounts/fireworks/models/flux-1-dev-controlnet-union", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-9" } ], "output": [ { "amount": 0.001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-9" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/flux-1-dev-fp8", "name": "fireworks_ai/accounts/fireworks/models/flux-1-dev-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.0005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "5e-10" } ], "output": [ { "amount": 0.0005, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "5e-10" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/flux-1-schnell", "name": "fireworks_ai/accounts/fireworks/models/flux-1-schnell", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/flux-1-schnell-fp8", "name": "fireworks_ai/accounts/fireworks/models/flux-1-schnell-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.00035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "3.5e-10" } ], "output": [ { "amount": 0.00035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "3.5e-10" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/gemma-2b-it", "name": "fireworks_ai/accounts/fireworks/models/gemma-2b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/gemma-3-27b-it", "name": "fireworks_ai/accounts/fireworks/models/gemma-3-27b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/gemma-7b", "name": "fireworks_ai/accounts/fireworks/models/gemma-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/gemma-7b-it", "name": "fireworks_ai/accounts/fireworks/models/gemma-7b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/gemma2-9b-it", "name": "fireworks_ai/accounts/fireworks/models/gemma2-9b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/glm-4p5v", "name": "fireworks_ai/accounts/fireworks/models/glm-4p5v", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/gpt-oss-safeguard-120b", "name": "fireworks_ai/accounts/fireworks/models/gpt-oss-safeguard-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/gpt-oss-safeguard-20b", "name": "fireworks_ai/accounts/fireworks/models/gpt-oss-safeguard-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/hermes-2-pro-mistral-7b", "name": "fireworks_ai/accounts/fireworks/models/hermes-2-pro-mistral-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/internvl3-38b", "name": "fireworks_ai/accounts/fireworks/models/internvl3-38b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/internvl3-78b", "name": "fireworks_ai/accounts/fireworks/models/internvl3-78b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/internvl3-8b", "name": "fireworks_ai/accounts/fireworks/models/internvl3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/japanese-stable-diffusion-xl", "name": "fireworks_ai/accounts/fireworks/models/japanese-stable-diffusion-xl", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.00013, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.3e-10" } ], "output": [ { "amount": 0.00013, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.3e-10" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/kat-coder", "name": "fireworks_ai/accounts/fireworks/models/kat-coder", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/kat-dev-32b", "name": "fireworks_ai/accounts/fireworks/models/kat-dev-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/kat-dev-72b-exp", "name": "fireworks_ai/accounts/fireworks/models/kat-dev-72b-exp", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-guard-2-8b", "name": "fireworks_ai/accounts/fireworks/models/llama-guard-2-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-guard-3-1b", "name": "fireworks_ai/accounts/fireworks/models/llama-guard-3-1b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-guard-3-8b", "name": "fireworks_ai/accounts/fireworks/models/llama-guard-3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v2-13b", "name": "fireworks_ai/accounts/fireworks/models/llama-v2-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v2-13b-chat", "name": "fireworks_ai/accounts/fireworks/models/llama-v2-13b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v2-70b", "name": "fireworks_ai/accounts/fireworks/models/llama-v2-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v2-70b-chat", "name": "fireworks_ai/accounts/fireworks/models/llama-v2-70b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 2048, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v2-7b", "name": "fireworks_ai/accounts/fireworks/models/llama-v2-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v2-7b-chat", "name": "fireworks_ai/accounts/fireworks/models/llama-v2-7b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3-70b-instruct", "name": "fireworks_ai/accounts/fireworks/models/llama-v3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3-70b-instruct-hf", "name": "fireworks_ai/accounts/fireworks/models/llama-v3-70b-instruct-hf", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3-8b", "name": "fireworks_ai/accounts/fireworks/models/llama-v3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3-8b-instruct-hf", "name": "fireworks_ai/accounts/fireworks/models/llama-v3-8b-instruct-hf", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p1-405b-instruct-long", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p1-405b-instruct-long", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p1-70b-instruct", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p1-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p1-70b-instruct-1b", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p1-70b-instruct-1b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p1-nemotron-70b-instruct", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p1-nemotron-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p2-1b", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p2-1b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p2-3b", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p2-3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct", "name": "fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llamaguard-7b", "name": "fireworks_ai/accounts/fireworks/models/llamaguard-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/llava-yi-34b", "name": "fireworks_ai/accounts/fireworks/models/llava-yi-34b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/minimax-m1-80k", "name": "fireworks_ai/accounts/fireworks/models/minimax-m1-80k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/minimax-m2", "name": "fireworks_ai/accounts/fireworks/models/minimax-m2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/ministral-3-14b-instruct-2512", "name": "fireworks_ai/accounts/fireworks/models/ministral-3-14b-instruct-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/ministral-3-3b-instruct-2512", "name": "fireworks_ai/accounts/fireworks/models/ministral-3-3b-instruct-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/ministral-3-8b-instruct-2512", "name": "fireworks_ai/accounts/fireworks/models/ministral-3-8b-instruct-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mistral-7b", "name": "fireworks_ai/accounts/fireworks/models/mistral-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-4k", "name": "fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-4k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-v0p2", "name": "fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-v0p2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-v3", "name": "fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mistral-7b-v0p2", "name": "fireworks_ai/accounts/fireworks/models/mistral-7b-v0p2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mistral-large-3-fp8", "name": "fireworks_ai/accounts/fireworks/models/mistral-large-3-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mistral-nemo-base-2407", "name": "fireworks_ai/accounts/fireworks/models/mistral-nemo-base-2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mistral-nemo-instruct-2407", "name": "fireworks_ai/accounts/fireworks/models/mistral-nemo-instruct-2407", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mistral-small-24b-instruct-2501", "name": "fireworks_ai/accounts/fireworks/models/mistral-small-24b-instruct-2501", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mixtral-8x22b", "name": "fireworks_ai/accounts/fireworks/models/mixtral-8x22b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 65536, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct", "name": "fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 65536, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mixtral-8x7b", "name": "fireworks_ai/accounts/fireworks/models/mixtral-8x7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mixtral-8x7b-instruct", "name": "fireworks_ai/accounts/fireworks/models/mixtral-8x7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mixtral-8x7b-instruct-hf", "name": "fireworks_ai/accounts/fireworks/models/mixtral-8x7b-instruct-hf", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/mythomax-l2-13b", "name": "fireworks_ai/accounts/fireworks/models/mythomax-l2-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/nemotron-nano-v2-12b-vl", "name": "fireworks_ai/accounts/fireworks/models/nemotron-nano-v2-12b-vl", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/nous-capybara-7b-v1p9", "name": "fireworks_ai/accounts/fireworks/models/nous-capybara-7b-v1p9", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/nous-hermes-2-mixtral-8x7b-dpo", "name": "fireworks_ai/accounts/fireworks/models/nous-hermes-2-mixtral-8x7b-dpo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/nous-hermes-2-yi-34b", "name": "fireworks_ai/accounts/fireworks/models/nous-hermes-2-yi-34b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-13b", "name": "fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-70b", "name": "fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-7b", "name": "fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/nvidia-nemotron-nano-12b-v2", "name": "fireworks_ai/accounts/fireworks/models/nvidia-nemotron-nano-12b-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/nvidia-nemotron-nano-9b-v2", "name": "fireworks_ai/accounts/fireworks/models/nvidia-nemotron-nano-9b-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/openchat-3p5-0106-7b", "name": "fireworks_ai/accounts/fireworks/models/openchat-3p5-0106-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/openhermes-2-mistral-7b", "name": "fireworks_ai/accounts/fireworks/models/openhermes-2-mistral-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/openhermes-2p5-mistral-7b", "name": "fireworks_ai/accounts/fireworks/models/openhermes-2p5-mistral-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/openorca-7b", "name": "fireworks_ai/accounts/fireworks/models/openorca-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/phi-2-3b", "name": "fireworks_ai/accounts/fireworks/models/phi-2-3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 2048, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/phi-3-mini-128k-instruct", "name": "fireworks_ai/accounts/fireworks/models/phi-3-mini-128k-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/phi-3-vision-128k-instruct", "name": "fireworks_ai/accounts/fireworks/models/phi-3-vision-128k-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32064, "maxOutputTokens": 32064, "maxTokens": 32064, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-python-v1", "name": "fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-python-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-v1", "name": "fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-v1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-v2", "name": "fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-v2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/playground-v2-1024px-aesthetic", "name": "fireworks_ai/accounts/fireworks/models/playground-v2-1024px-aesthetic", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.00013, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.3e-10" } ], "output": [ { "amount": 0.00013, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.3e-10" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/playground-v2-5-1024px-aesthetic", "name": "fireworks_ai/accounts/fireworks/models/playground-v2-5-1024px-aesthetic", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.00013, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.3e-10" } ], "output": [ { "amount": 0.00013, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.3e-10" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/pythia-12b", "name": "fireworks_ai/accounts/fireworks/models/pythia-12b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 2048, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen-qwq-32b-preview", "name": "fireworks_ai/accounts/fireworks/models/qwen-qwq-32b-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen-v2p5-14b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen-v2p5-14b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen-v2p5-7b", "name": "fireworks_ai/accounts/fireworks/models/qwen-v2p5-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen1p5-72b-chat", "name": "fireworks_ai/accounts/fireworks/models/qwen1p5-72b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2-7b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2-7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2-vl-2b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2-vl-2b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2-vl-72b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2-vl-72b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2-vl-7b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2-vl-7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-0p5b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-0p5b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-14b", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-14b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-1p5b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-1p5b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-32b", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-32b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-32b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-72b", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-72b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-72b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-72b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-7b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-0p5b", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-0p5b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-0p5b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-0p5b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-14b", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-14b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-14b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-14b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-1p5b", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-1p5b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-1p5b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-1p5b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-128k", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-128k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-32k-rope", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-32k-rope", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-64k", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-64k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 65536, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-3b", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-3b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-7b", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-7b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-coder-7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-math-72b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-math-72b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-32b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-32b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-3b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-72b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-72b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-7b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen2p5-vl-7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-0p6b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-0p6b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-14b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-14b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-1p7b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-1p7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft", "name": "fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft-131072", "name": "fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft-131072", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft-40960", "name": "fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft-40960", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b-instruct-2507", "name": "fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b-instruct-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b-thinking-2507", "name": "fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b-thinking-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b-instruct-2507", "name": "fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b-instruct-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b-thinking-2507", "name": "fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b-thinking-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-32b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-4b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-4b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-4b-instruct-2507", "name": "fireworks_ai/accounts/fireworks/models/qwen3-4b-instruct-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-8b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-coder-30b-a3b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen3-coder-30b-a3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-coder-480b-instruct-bf16", "name": "fireworks_ai/accounts/fireworks/models/qwen3-coder-480b-instruct-bf16", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-embedding-0p6b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-embedding-0p6b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-embedding-4b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-embedding-4b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "fireworks_ai", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/", "name": "fireworks_ai/accounts/fireworks/models/", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "fireworks_ai", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-next-80b-a3b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen3-next-80b-a3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-next-80b-a3b-thinking", "name": "fireworks_ai/accounts/fireworks/models/qwen3-next-80b-a3b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-reranker-0p6b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-reranker-0p6b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "fireworks_ai", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-reranker-4b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-reranker-4b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "fireworks_ai", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-reranker-8b", "name": "fireworks_ai/accounts/fireworks/models/qwen3-reranker-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "fireworks_ai", "maxInputTokens": 40960, "maxOutputTokens": 40960, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-vl-235b-a22b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen3-vl-235b-a22b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-vl-235b-a22b-thinking", "name": "fireworks_ai/accounts/fireworks/models/qwen3-vl-235b-a22b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" } ], "output": [ { "amount": 0.88, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-vl-30b-a3b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen3-vl-30b-a3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-vl-30b-a3b-thinking", "name": "fireworks_ai/accounts/fireworks/models/qwen3-vl-30b-a3b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-vl-32b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen3-vl-32b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3-vl-8b-instruct", "name": "fireworks_ai/accounts/fireworks/models/qwen3-vl-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwen3p7-plus", "name": "fireworks_ai/accounts/fireworks/models/qwen3p7-plus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.5999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000016" } ], "cacheRead": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/qwq-32b", "name": "fireworks_ai/accounts/fireworks/models/qwq-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/rolm-ocr", "name": "fireworks_ai/accounts/fireworks/models/rolm-ocr", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/snorkel-mistral-7b-pairrm-dpo", "name": "fireworks_ai/accounts/fireworks/models/snorkel-mistral-7b-pairrm-dpo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/stable-diffusion-xl-1024-v1-0", "name": "fireworks_ai/accounts/fireworks/models/stable-diffusion-xl-1024-v1-0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.00013, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.3e-10" } ], "output": [ { "amount": 0.00013, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "1.3e-10" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/stablecode-3b", "name": "fireworks_ai/accounts/fireworks/models/stablecode-3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/starcoder-16b", "name": "fireworks_ai/accounts/fireworks/models/starcoder-16b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/starcoder-7b", "name": "fireworks_ai/accounts/fireworks/models/starcoder-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/starcoder2-15b", "name": "fireworks_ai/accounts/fireworks/models/starcoder2-15b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/starcoder2-3b", "name": "fireworks_ai/accounts/fireworks/models/starcoder2-3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/starcoder2-7b", "name": "fireworks_ai/accounts/fireworks/models/starcoder2-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/toppy-m-7b", "name": "fireworks_ai/accounts/fireworks/models/toppy-m-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/yi-34b", "name": "fireworks_ai/accounts/fireworks/models/yi-34b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/yi-34b-200k-capybara", "name": "fireworks_ai/accounts/fireworks/models/yi-34b-200k-capybara", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 200000, "maxOutputTokens": 200000, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/yi-34b-chat", "name": "fireworks_ai/accounts/fireworks/models/yi-34b-chat", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/yi-6b", "name": "fireworks_ai/accounts/fireworks/models/yi-6b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 4096, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/models/zephyr-7b-beta", "name": "fireworks_ai/accounts/fireworks/models/zephyr-7b-beta", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/routers/glm-5p1-fast", "name": "fireworks_ai/accounts/fireworks/routers/glm-5p1-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000028" } ], "output": [ { "amount": 8.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000088" } ], "cacheRead": [ { "amount": 0.52, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.2e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 202800, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/routers/kimi-k2p6-fast", "name": "fireworks_ai/accounts/fireworks/routers/kimi-k2p6-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "fireworks_ai/accounts/fireworks/routers/kimi-k2p7-code-fast", "name": "fireworks_ai/accounts/fireworks/routers/kimi-k2p7-code-fast", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.9, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000019" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "fireworks_ai", "maxInputTokens": 262144, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/qwen/qwen3.5-397b-a17b", "name": "scaleway/qwen/qwen3.5-397b-a17b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 256000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/qwen/qwen3.6-35b-a3b", "name": "scaleway/qwen/qwen3.6-35b-a3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 256000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/qwen/qwen3-235b-a22b-instruct-2507", "name": "scaleway/qwen/qwen3-235b-a22b-instruct-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "output": [ { "amount": 2.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000225" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 256000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/qwen/qwen3-embedding-8b", "name": "scaleway/qwen/qwen3-embedding-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "scaleway", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/qwen/qwen3-coder-30b-a3b-instruct", "name": "scaleway/qwen/qwen3-coder-30b-a3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 128000, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/openai/gpt-oss-120b", "name": "scaleway/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 128000, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/openai/whisper-large-v3", "name": "scaleway/openai/whisper-large-v3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "scaleway", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/google/gemma-4-26b-a4b-it", "name": "scaleway/google/gemma-4-26b-a4b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 256000, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/google/gemma-3-27b-it", "name": "scaleway/google/gemma-3-27b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 40000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/hcompany/holo2-30b-a3b", "name": "scaleway/hcompany/holo2-30b-a3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 22000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/mistralai/mistral-medium-3.5-128b", "name": "scaleway/mistralai/mistral-medium-3.5-128b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "output": [ { "amount": 7.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000075" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 256000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/mistralai/devstral-2-123b-instruct-2512", "name": "scaleway/mistralai/devstral-2-123b-instruct-2512", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 200000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/mistralai/voxtral-small-24b-2507", "name": "scaleway/mistralai/voxtral-small-24b-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" }, { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "1.5e-7" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 32000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/mistralai/mistral-small-3.2-24b-instruct-2506", "name": "scaleway/mistralai/mistral-small-3.2-24b-instruct-2506", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 128000, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/mistralai/pixtral-12b-2409", "name": "scaleway/mistralai/pixtral-12b-2409", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/BAAI/bge-multilingual-gemma2", "name": "scaleway/BAAI/bge-multilingual-gemma2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-7" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "scaleway", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "scaleway/meta/llama-3.3-70b-instruct", "name": "scaleway/meta/llama-3.3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "scaleway", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-v3.2", "name": "novita/deepseek/deepseek-v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.26899999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.69e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.13449999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.345e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 163840, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/minimax/minimax-m2.1", "name": "novita/minimax/minimax-m2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 204800, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/zai-org/glm-4.7", "name": "novita/zai-org/glm-4.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 204800, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/xiaomimimo/mimo-v2-flash", "name": "novita/xiaomimimo/mimo-v2-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 262144, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/zai-org/autoglm-phone-9b-multilingual", "name": "novita/zai-org/autoglm-phone-9b-multilingual", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-8" } ], "output": [ { "amount": 0.13799999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.38e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 65536, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/moonshotai/kimi-k2-thinking", "name": "novita/moonshotai/kimi-k2-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/minimax/minimax-m2", "name": "novita/minimax/minimax-m2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 204800, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/paddlepaddle/paddleocr-vl", "name": "novita/paddlepaddle/paddleocr-vl", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "output": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-v3.2-exp", "name": "novita/deepseek/deepseek-v3.2-exp", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 0.41, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 163840, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-vl-235b-a22b-thinking", "name": "novita/qwen/qwen3-vl-235b-a22b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.98, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.8e-7" } ], "output": [ { "amount": 3.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000395" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/zai-org/glm-4.6v", "name": "novita/zai-org/glm-4.6v", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.8999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-7" } ], "cacheRead": [ { "amount": 0.055, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/zai-org/glm-4.6", "name": "novita/zai-org/glm-4.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 204800, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/kwaipilot/kat-coder-pro", "name": "novita/kwaipilot/kat-coder-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 256000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-next-80b-a3b-instruct", "name": "novita/qwen/qwen3-next-80b-a3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-next-80b-a3b-thinking", "name": "novita/qwen/qwen3-next-80b-a3b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-ocr", "name": "novita/deepseek/deepseek-ocr", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-v3.1-terminus", "name": "novita/deepseek/deepseek-v3.1-terminus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.35e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-vl-235b-a22b-instruct", "name": "novita/qwen/qwen3-vl-235b-a22b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-max", "name": "novita/qwen/qwen3-max", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.1100000000000003, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000211" } ], "output": [ { "amount": 8.450000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000845" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/skywork/r1v4-lite", "name": "novita/skywork/r1v4-lite", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-v3.1", "name": "novita/deepseek/deepseek-v3.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.35e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/moonshotai/kimi-k2-0905", "name": "novita/moonshotai/kimi-k2-0905", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-coder-480b-a35b-instruct", "name": "novita/qwen/qwen3-coder-480b-a35b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000013" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 262144, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-coder-30b-a3b-instruct", "name": "novita/qwen/qwen3-coder-30b-a3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 160000, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/openai/gpt-oss-120b", "name": "novita/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/moonshotai/kimi-k2-instruct", "name": "novita/moonshotai/kimi-k2-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.5700000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.7e-7" } ], "output": [ { "amount": 2.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000023" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-v3-0324", "name": "novita/deepseek/deepseek-v3-0324", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 1.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000112" } ], "cacheRead": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.35e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/zai-org/glm-4.5", "name": "novita/zai-org/glm-4.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 98304, "maxTokens": 98304, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-235b-a22b-thinking-2507", "name": "novita/qwen/qwen3-235b-a22b-thinking-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/meta-llama/llama-3.1-8b-instruct", "name": "novita/meta-llama/llama-3.1-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-8" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 16384, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/google/gemma-3-12b-it", "name": "novita/google/gemma-3-12b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/zai-org/glm-4.5v", "name": "novita/zai-org/glm-4.5v", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [ { "amount": 0.11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 65536, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/openai/gpt-oss-20b", "name": "novita/openai/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-235b-a22b-instruct-2507", "name": "novita/qwen/qwen3-235b-a22b-instruct-2507", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.58, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-r1-distill-qwen-14b", "name": "novita/deepseek/deepseek-r1-distill-qwen-14b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 32768, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/meta-llama/llama-3.3-70b-instruct", "name": "novita/meta-llama/llama-3.3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.135, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.35e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 120000, "maxTokens": 120000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen-2.5-72b-instruct", "name": "novita/qwen/qwen-2.5-72b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.38, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.8e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 32000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/mistralai/mistral-nemo", "name": "novita/mistralai/mistral-nemo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.16999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 60288, "maxOutputTokens": 16000, "maxTokens": 16000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/minimaxai/minimax-m1-80k", "name": "novita/minimaxai/minimax-m1-80k", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 1000000, "maxOutputTokens": 40000, "maxTokens": 40000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-r1-0528", "name": "novita/deepseek/deepseek-r1-0528", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.35, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 163840, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-r1-distill-qwen-32b", "name": "novita/deepseek/deepseek-r1-distill-qwen-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 64000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/meta-llama/llama-3-8b-instruct", "name": "novita/meta-llama/llama-3-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/microsoft/wizardlm-2-8x22b", "name": "novita/microsoft/wizardlm-2-8x22b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.62, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.2e-7" } ], "output": [ { "amount": 0.62, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 65535, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-r1-0528-qwen3-8b", "name": "novita/deepseek/deepseek-r1-0528-qwen3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 128000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-r1-distill-llama-70b", "name": "novita/deepseek/deepseek-r1-distill-llama-70b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/meta-llama/llama-3-70b-instruct", "name": "novita/meta-llama/llama-3-70b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.51, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.1e-7" } ], "output": [ { "amount": 0.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 8192, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-235b-a22b-fp8", "name": "novita/qwen/qwen3-235b-a22b-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 40960, "maxOutputTokens": 20000, "maxTokens": 20000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/meta-llama/llama-4-maverick-17b-128e-instruct-fp8", "name": "novita/meta-llama/llama-4-maverick-17b-128e-instruct-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.27, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.7e-7" } ], "output": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/meta-llama/llama-4-scout-17b-16e-instruct", "name": "novita/meta-llama/llama-4-scout-17b-16e-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.18, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.8e-7" } ], "output": [ { "amount": 0.59, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/nousresearch/hermes-2-pro-llama-3-8b", "name": "novita/nousresearch/hermes-2-pro-llama-3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen2.5-vl-72b-instruct", "name": "novita/qwen/qwen2.5-vl-72b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "output": [ { "amount": 0.7999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/sao10k/l3-70b-euryale-v2.1", "name": "novita/sao10k/l3-70b-euryale-v2.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000148" } ], "output": [ { "amount": 1.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000148" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/baidu/ernie-4.5-21B-a3b-thinking", "name": "novita/baidu/ernie-4.5-21B-a3b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/sao10k/l3-8b-lunaris", "name": "novita/sao10k/l3-8b-lunaris", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/baichuan/baichuan-m2-32b", "name": "novita/baichuan/baichuan-m2-32b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/baidu/ernie-4.5-vl-424b-a47b", "name": "novita/baidu/ernie-4.5-vl-424b-a47b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.42, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.2e-7" } ], "output": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 123000, "maxOutputTokens": 16000, "maxTokens": 16000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/baidu/ernie-4.5-300b-a47b-paddle", "name": "novita/baidu/ernie-4.5-300b-a47b-paddle", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "output": [ { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 123000, "maxOutputTokens": 12000, "maxTokens": 12000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-prover-v2-671b", "name": "novita/deepseek/deepseek-prover-v2-671b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 160000, "maxOutputTokens": 160000, "maxTokens": 160000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-32b-fp8", "name": "novita/qwen/qwen3-32b-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 40960, "maxOutputTokens": 20000, "maxTokens": 20000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-30b-a3b-fp8", "name": "novita/qwen/qwen3-30b-a3b-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 40960, "maxOutputTokens": 20000, "maxTokens": 20000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/google/gemma-3-27b-it", "name": "novita/google/gemma-3-27b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.119, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.19e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 98304, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-v3-turbo", "name": "novita/deepseek/deepseek-v3-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "output": [ { "amount": 1.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000013" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 64000, "maxOutputTokens": 16000, "maxTokens": 16000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/deepseek/deepseek-r1-turbo", "name": "novita/deepseek/deepseek-r1-turbo", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 64000, "maxOutputTokens": 16000, "maxTokens": 16000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/Sao10K/L3-8B-Stheno-v3.2", "name": "novita/Sao10K/L3-8B-Stheno-v3.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 8192, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/gryphe/mythomax-l2-13b", "name": "novita/gryphe/mythomax-l2-13b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 4096, "maxOutputTokens": 3200, "maxTokens": 3200, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/baidu/ernie-4.5-vl-28b-a3b-thinking", "name": "novita/baidu/ernie-4.5-vl-28b-a3b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.39, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.9e-7" } ], "output": [ { "amount": 0.39, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.9e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-vl-8b-instruct", "name": "novita/qwen/qwen3-vl-8b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/zai-org/glm-4.5-air", "name": "novita/zai-org/glm-4.5-air", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.85, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 98304, "maxTokens": 98304, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-vl-30b-a3b-instruct", "name": "novita/qwen/qwen3-vl-30b-a3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 0.7, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-vl-30b-a3b-thinking", "name": "novita/qwen/qwen3-vl-30b-a3b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "output": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-omni-30b-a3b-thinking", "name": "novita/qwen/qwen3-omni-30b-a3b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.9700000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 65536, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-omni-30b-a3b-instruct", "name": "novita/qwen/qwen3-omni-30b-a3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.9700000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 65536, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen-mt-plus", "name": "novita/qwen/qwen-mt-plus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 16384, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/baidu/ernie-4.5-vl-28b-a3b", "name": "novita/baidu/ernie-4.5-vl-28b-a3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 30000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/baidu/ernie-4.5-21B-a3b", "name": "novita/baidu/ernie-4.5-21B-a3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 120000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-8b-fp8", "name": "novita/qwen/qwen3-8b-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.035, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.5e-8" } ], "output": [ { "amount": 0.13799999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.38e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 128000, "maxOutputTokens": 20000, "maxTokens": 20000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-4b-fp8", "name": "novita/qwen/qwen3-4b-fp8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "output": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 128000, "maxOutputTokens": 20000, "maxTokens": 20000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen2.5-7b-instruct", "name": "novita/qwen/qwen2.5-7b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 32000, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "novita/meta-llama/llama-3.2-3b-instruct", "name": "novita/meta-llama/llama-3.2-3b-instruct", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 32768, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/sao10k/l31-70b-euryale-v2.2", "name": "novita/sao10k/l31-70b-euryale-v2.2", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000148" } ], "output": [ { "amount": 1.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000148" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "novita", "maxInputTokens": 8192, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "novita/qwen/qwen3-embedding-0.6b", "name": "novita/qwen/qwen3-embedding-0.6b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "7e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "novita", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "novita/qwen/qwen3-embedding-8b", "name": "novita/qwen/qwen3-embedding-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "7e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "novita", "maxInputTokens": 32768, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "novita/baai/bge-m3", "name": "novita/baai/bge-m3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-8" } ], "output": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "novita", "maxInputTokens": 8192, "maxOutputTokens": 96000, "maxTokens": 96000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "novita/qwen/qwen3-reranker-8b", "name": "novita/qwen/qwen3-reranker-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "5e-8" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "novita", "maxInputTokens": 32768, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "novita/baai/bge-reranker-v2-m3", "name": "novita/baai/bge-reranker-v2-m3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "1e-8" } ], "output": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "rerank", "raw": "1e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "rerank", "servingProvider": "novita", "maxInputTokens": 8000, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/llama-3.1-8b", "name": "llamagate/llama-3.1-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "output": [ { "amount": 0.049999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/llama-3.2-3b", "name": "llamagate/llama-3.2-3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 131072, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/mistral-7b-v0.3", "name": "llamagate/mistral-7b-v0.3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 32768, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/qwen3-8b", "name": "llamagate/qwen3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 32768, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/dolphin3-8b", "name": "llamagate/dolphin3-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/deepseek-r1-8b", "name": "llamagate/deepseek-r1-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 65536, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/deepseek-r1-7b-qwen", "name": "llamagate/deepseek-r1-7b-qwen", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 131072, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/openthinker-7b", "name": "llamagate/openthinker-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "output": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 32768, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/qwen2.5-coder-7b", "name": "llamagate/qwen2.5-coder-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 32768, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/deepseek-coder-6.7b", "name": "llamagate/deepseek-coder-6.7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 16384, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/codellama-7b", "name": "llamagate/codellama-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-8" } ], "output": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 16384, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/qwen3-vl-8b", "name": "llamagate/qwen3-vl-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 32768, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/llava-7b", "name": "llamagate/llava-7b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 4096, "maxOutputTokens": 2048, "maxTokens": 2048, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/gemma3-4b", "name": "llamagate/gemma3-4b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "llamagate", "maxInputTokens": 128000, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/nomic-embed-text", "name": "llamagate/nomic-embed-text", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "llamagate", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "llamagate/qwen3-embedding-8b", "name": "llamagate/qwen3-embedding-8b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.02, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "2e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "llamagate", "maxInputTokens": 40960, "maxOutputTokens": null, "maxTokens": 40960, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "libertai/hermes-3-8b-tee", "name": "libertai/hermes-3-8b-tee", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "libertai", "maxInputTokens": 16000, "maxOutputTokens": 16000, "maxTokens": 16000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "libertai/gemma-4-31b-it", "name": "libertai/gemma-4-31b-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "libertai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "libertai/gemma-4-31b-it-thinking", "name": "libertai/gemma-4-31b-it-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "libertai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "libertai/qwen3.6-27b", "name": "libertai/qwen3.6-27b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "libertai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "libertai/qwen3.6-27b-thinking", "name": "libertai/qwen3.6-27b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "libertai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "libertai/qwen3.6-35b-a3b", "name": "libertai/qwen3.6-35b-a3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "libertai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "libertai/qwen3.6-35b-a3b-thinking", "name": "libertai/qwen3.6-35b-a3b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "libertai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "libertai/qwen3.5-122b-a10b", "name": "libertai/qwen3.5-122b-a10b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "libertai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "libertai/qwen3.5-122b-a10b-thinking", "name": "libertai/qwen3.5-122b-a10b-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "libertai", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "libertai/deepseek-v4-flash", "name": "libertai/deepseek-v4-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "libertai", "maxInputTokens": 200000, "maxOutputTokens": 200000, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "libertai/deepseek-v4-flash-thinking", "name": "libertai/deepseek-v4-flash-thinking", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "output": [ { "amount": 1.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000175" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "libertai", "maxInputTokens": 200000, "maxOutputTokens": 200000, "maxTokens": 200000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "libertai/bge-m3", "name": "libertai/bge-m3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "1e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "libertai", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "sarvam/sarvam-m", "name": "sarvam/sarvam-m", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" }, { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "sarvam", "maxInputTokens": 8192, "maxOutputTokens": 32000, "maxTokens": 32000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "tts-1-1106", "name": "tts-1-1106", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000015, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.000015" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "tts-1-hd-1106", "name": "tts-1-hd-1106", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00003, "currency": "USD", "units": 1, "pricingType": "character", "raw": "0.00003" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-4o-mini-tts-2025-03-20", "name": "gpt-4o-mini-tts-2025-03-20", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00025, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00025" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-07-23", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-4o-mini-tts-2025-12-15", "name": "gpt-4o-mini-tts-2025-12-15", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00025, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00025" } ] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-4o-mini-transcribe-2025-03-20", "name": "gpt-4o-mini-transcribe-2025-03-20", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000125" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000125" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "openai", "maxInputTokens": 16000, "maxOutputTokens": 2000, "maxTokens": null, "deprecationDate": "2027-01-20", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-4o-mini-transcribe-2025-12-15", "name": "gpt-4o-mini-transcribe-2025-12-15", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000125" }, { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00000125" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000005" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "openai", "maxInputTokens": 16000, "maxOutputTokens": 2000, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-5-search-api", "name": "gpt-5-search-api", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-5-search-api-2025-10-14", "name": "gpt-5-search-api-2025-10-14", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "openai", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gpt-realtime-mini-2025-10-06", "name": "gpt-realtime-mini-2025-10-06", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-8" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "other": [ { "amount": 8e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "8e-7" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": "2026-07-23", "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-realtime-mini-2025-12-15", "name": "gpt-realtime-mini-2025-12-15", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-7" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00001" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000024" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [ { "amount": 0.06, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "6e-8" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "other": [ { "amount": 8e-7, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "8e-7" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 128000, "maxOutputTokens": 4096, "maxTokens": 4096, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "gpt-realtime-whisper", "name": "gpt-realtime-whisper", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0002833333333333333, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0002833333333333333" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "chatgpt-image-latest", "name": "chatgpt-image-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000005" }, { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00001" } ], "output": [ { "amount": 40, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00004" } ], "cacheRead": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.00000125" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": "2026-12-01", "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini-2.0-flash-exp-image-generation", "name": "gemini-2.0-flash-exp-image-generation", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.039, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.039" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-2.0-flash-exp-image-generation", "name": "gemini/gemini-2.0-flash-exp-image-generation", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.039, "currency": "USD", "units": 1, "pricingType": "image", "modality": "image", "raw": "0.039" } ] } ], "metadata": { "source": "litellm", "mode": "image_generation", "servingProvider": "gemini", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-2.0-flash-lite-001", "name": "gemini/gemini-2.0-flash-lite-001", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" }, { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [ { "amount": 0.01875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.875e-8" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": null, "deprecationDate": "2026-06-01", "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": true, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-2.5-flash-native-audio-latest", "name": "gemini-2.5-flash-native-audio-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini-2.5-flash-native-audio-preview-09-2025", "name": "gemini-2.5-flash-native-audio-preview-09-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini-2.5-flash-native-audio-preview-12-2025", "name": "gemini-2.5-flash-native-audio-preview-12-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini-3.1-flash-live-preview", "name": "gemini-3.1-flash-live-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000003" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000033333333333333335, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.000033333333333333335" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "gemini/gemini-2.5-flash-native-audio-latest", "name": "gemini/gemini-2.5-flash-native-audio-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-2.5-flash-native-audio-preview-09-2025", "name": "gemini/gemini-2.5-flash-native-audio-preview-09-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-2.5-flash-native-audio-preview-12-2025", "name": "gemini/gemini-2.5-flash-native-audio-preview-12-2025", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 8192, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-3.1-flash-live-preview", "name": "gemini/gemini-3.1-flash-live-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-7" }, { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000003" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "image", "raw": "0.000001" } ], "output": [ { "amount": 4.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000045" }, { "amount": 12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000012" } ], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000033333333333333335, "currency": "USD", "units": 1, "pricingType": "second", "modality": "video", "raw": "0.000033333333333333335" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "gemini/gemini-3.1-flash-tts-preview", "name": "gemini/gemini-3.1-flash-tts-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.00002" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "gemini", "maxInputTokens": 8192, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini-2.5-flash-preview-tts", "name": "gemini-2.5-flash-preview-tts", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.0000025" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "audio_speech", "servingProvider": "gemini", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gemini-flash-latest", "name": "gemini-flash-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-flash-lite-latest", "name": "gemini-flash-lite-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" }, { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [ { "amount": 0.01, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-8" } ], "cacheWrite": [], "other": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-pro-latest", "name": "gemini-pro-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini/gemini-pro-latest", "name": "gemini/gemini-pro-latest", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" }, { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" }, { "amount": 0.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "gemini-exp-1206", "name": "gemini-exp-1206", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" }, { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000001" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "cacheWrite": [], "other": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": 1048576, "maxOutputTokens": 65535, "maxTokens": 65535, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": true, "audioInput": null, "audioOutput": false, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": true } } }, { "id": "vertex_ai/claude-sonnet-5@default", "name": "vertex_ai/claude-sonnet-5@default", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "vertex_ai/claude-sonnet-4-6@default", "name": "vertex_ai/claude-sonnet-4-6@default", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [ { "amount": 3.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000375" }, { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "vertex_ai-anthropic_models", "maxInputTokens": 1000000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "duckduckgo/search", "name": "duckduckgo/search", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "request", "raw": "0" } ] } ], "metadata": { "source": "litellm", "mode": "search", "servingProvider": "duckduckgo", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/openai.gpt-oss-120b", "name": "bedrock_mantle/openai.gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_mantle", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/openai.gpt-oss-20b", "name": "bedrock_mantle/openai.gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_mantle", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/openai.gpt-oss-safeguard-120b", "name": "bedrock_mantle/openai.gpt-oss-safeguard-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_mantle", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/openai.gpt-oss-safeguard-20b", "name": "bedrock_mantle/openai.gpt-oss-safeguard-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.075, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7.5e-8" } ], "output": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_mantle", "maxInputTokens": 131072, "maxOutputTokens": 65536, "maxTokens": 65536, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/openai.gpt-5.6-sol", "name": "bedrock_mantle/openai.gpt-5.6-sol", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" }, { "amount": 11, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000011" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" }, { "amount": 49.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000495" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" }, { "amount": 1.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000011" } ], "cacheWrite": [ { "amount": 6.875, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006875" }, { "amount": 13.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001375" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "bedrock_mantle", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/openai.gpt-5.6-terra", "name": "bedrock_mantle/openai.gpt-5.6-terra", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000022" }, { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "output": [ { "amount": 13.200000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000132" }, { "amount": 19.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000198" } ], "cacheRead": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" }, { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-7" } ], "cacheWrite": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" }, { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "bedrock_mantle", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/openai.gpt-5.6-luna", "name": "bedrock_mantle/openai.gpt-5.6-luna", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.22, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-7" }, { "amount": 0.44, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-7" } ], "output": [ { "amount": 1.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000132" }, { "amount": 1.9800000000000002, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000198" } ], "cacheRead": [ { "amount": 0.022, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2e-8" }, { "amount": 0.044, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4e-8" } ], "cacheWrite": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-7" }, { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "bedrock_mantle", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/openai.gpt-5.5", "name": "bedrock_mantle/openai.gpt-5.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000055" } ], "output": [ { "amount": 33, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000033" } ], "cacheRead": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "bedrock_mantle", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/openai.gpt-5.4", "name": "bedrock_mantle/openai.gpt-5.4", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2.75, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000275" } ], "output": [ { "amount": 16.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000165" } ], "cacheRead": [ { "amount": 0.275, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.75e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "responses", "servingProvider": "bedrock_mantle", "maxInputTokens": 272000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/google.gemma-4-31b", "name": "bedrock_mantle/google.gemma-4-31b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_mantle", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/google.gemma-4-26b-a4b", "name": "bedrock_mantle/google.gemma-4-26b-a4b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.3e-7" } ], "output": [ { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_mantle", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/google.gemma-4-e2b", "name": "bedrock_mantle/google.gemma-4-e2b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.04, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-8" } ], "output": [ { "amount": 0.08, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_mantle", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": false, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock_mantle/xai.grok-4.3", "name": "bedrock_mantle/xai.grok-4.3", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock_mantle", "maxInputTokens": 131072, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-east-1/zai.glm-5", "name": "bedrock/us-east-1/zai.glm-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-west-2/zai.glm-5", "name": "bedrock/us-west-2/zai.glm-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0", "name": "bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "cacheWrite": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "bedrock/us-gov-west-1/anthropic.claude-haiku-4-5-20251001-v1:0", "name": "bedrock/us-gov-west-1/anthropic.claude-haiku-4-5-20251001-v1:0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "output": [ { "amount": 6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000006" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2e-7" } ], "cacheWrite": [ { "amount": 1.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000015" }, { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000024" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "bedrock", "maxInputTokens": 200000, "maxOutputTokens": 64000, "maxTokens": 64000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "snowflake/claude-sonnet-4-5", "name": "snowflake/claude-sonnet-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 200000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/claude-sonnet-4-6", "name": "snowflake/claude-sonnet-4-6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 200000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/claude-4-sonnet", "name": "snowflake/claude-4-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 200000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/claude-4-opus", "name": "snowflake/claude-4-opus", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "output": [ { "amount": 25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000025" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 200000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/claude-haiku-4-5", "name": "snowflake/claude-haiku-4-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "output": [ { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 200000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/claude-3-7-sonnet", "name": "snowflake/claude-3-7-sonnet", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000003" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000015" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 200000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/openai-gpt-4.1", "name": "snowflake/openai-gpt-4.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000008" } ], "cacheRead": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 300000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/openai-gpt-5", "name": "snowflake/openai-gpt-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 300000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/openai-gpt-5-mini", "name": "snowflake/openai-gpt-5-mini", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 1000000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/openai-gpt-5-nano", "name": "snowflake/openai-gpt-5-nano", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 5000000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/llama4-maverick", "name": "snowflake/llama4-maverick", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.24, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4e-7" } ], "output": [ { "amount": 0.9700000000000001, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.7e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "snowflake", "maxInputTokens": 128000, "maxOutputTokens": 16384, "maxTokens": 16384, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "snowflake/snowflake-arctic-embed-l-v2.0", "name": "snowflake/snowflake-arctic-embed-l-v2.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "7e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "snowflake", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "snowflake/snowflake-arctic-embed-m-v2.0", "name": "snowflake/snowflake-arctic-embed-m-v2.0", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "7e-8" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "embedding", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "embedding", "servingProvider": "snowflake", "maxInputTokens": 8192, "maxOutputTokens": null, "maxTokens": 8192, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "soniox/stt-async-v4", "name": "soniox/stt-async-v4", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" }, { "amount": 0.0000277778, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0000277778" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "soniox", "maxInputTokens": null, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "soniox/stt-async-v5", "name": "soniox/stt-async-v5", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0" }, { "amount": 0.0000277778, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0000277778" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "soniox", "maxInputTokens": null, "maxOutputTokens": 8000, "maxTokens": 8000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8", "name": "tensormesh/Qwen/Qwen3.5-397B-A17B-FP8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "output": [ { "amount": 3.5999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000036" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tensormesh", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8", "name": "tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "output": [ { "amount": 1.7999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000018" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tensormesh", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "tensormesh/Qwen/Qwen3.6-27B-FP8", "name": "tensormesh/Qwen/Qwen3.6-27B-FP8", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.32, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.2e-7" } ], "output": [ { "amount": 3.1999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000032" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tensormesh", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "tensormesh/lukealonso/GLM-5.1-NVFP4-MTP", "name": "tensormesh/lukealonso/GLM-5.1-NVFP4-MTP", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000014" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000044" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tensormesh", "maxInputTokens": 202752, "maxOutputTokens": 202752, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "tensormesh/deepseek-ai/DeepSeek-V4-Flash", "name": "tensormesh/deepseek-ai/DeepSeek-V4-Flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tensormesh", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "tensormesh/moonshotai/Kimi-K2.6", "name": "tensormesh/moonshotai/Kimi-K2.6", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.96, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9.6e-7" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tensormesh", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "tensormesh/MiniMaxAI/MiniMax-M2.5", "name": "tensormesh/MiniMaxAI/MiniMax-M2.5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000012" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tensormesh", "maxInputTokens": 196608, "maxOutputTokens": 196608, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "tensormesh/google/gemma-4-31B-it", "name": "tensormesh/google/gemma-4-31B-it", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.6e-7" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tensormesh", "maxInputTokens": 32768, "maxOutputTokens": 32768, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "tensormesh/openai/gpt-oss-120b", "name": "tensormesh/openai/gpt-oss-120b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tensormesh", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "tensormesh/openai/gpt-oss-20b", "name": "tensormesh/openai/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tensormesh", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "deepseek-v4-flash", "name": "deepseek-v4-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [ { "amount": 0.0028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-9" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 1000000, "maxOutputTokens": 393216, "maxTokens": 393216, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "deepseek-v4-pro", "name": "deepseek-v4-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.435, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.35e-7" } ], "output": [ { "amount": 0.87, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.7e-7" } ], "cacheRead": [ { "amount": 0.003625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.625e-9" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 1000000, "maxOutputTokens": 393216, "maxTokens": 393216, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "deepseek/deepseek-v4-flash", "name": "deepseek/deepseek-v4-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [ { "amount": 0.0028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-9" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 1000000, "maxOutputTokens": 393216, "maxTokens": 393216, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "deepseek/deepseek-v4-pro", "name": "deepseek/deepseek-v4-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.435, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.35e-7" } ], "output": [ { "amount": 0.87, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.7e-7" } ], "cacheRead": [ { "amount": 0.003625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.625e-9" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "deepseek", "maxInputTokens": 1000000, "maxOutputTokens": 393216, "maxTokens": 393216, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "tencent/deepseek-v4-pro", "name": "tencent/deepseek-v4-pro", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.435, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.35e-7" } ], "output": [ { "amount": 0.87, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "8.7e-7" } ], "cacheRead": [ { "amount": 0.003625, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.625e-9" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tencent", "maxInputTokens": 1000000, "maxOutputTokens": 384000, "maxTokens": 384000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "tencent/deepseek-v4-flash", "name": "tencent/deepseek-v4-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-7" } ], "cacheRead": [ { "amount": 0.0028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.8e-9" } ], "cacheWrite": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "tencent", "maxInputTokens": 1000000, "maxOutputTokens": 384000, "maxTokens": 384000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": true, "vision": false, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": true, "webSearch": null } } }, { "id": "pinstripes/ps/glm-4.5-air", "name": "pinstripes/ps/glm-4.5-air", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.125, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.25e-7" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "pinstripes", "maxInputTokens": 128000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "pinstripes/ps/qwen3.6-35b-a3b", "name": "pinstripes/ps/qwen3.6-35b-a3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4e-7" } ], "output": [ { "amount": 0.44999999999999996, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "pinstripes", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "pinstripes/ps/qwen3-30b-a3b", "name": "pinstripes/ps/qwen3-30b-a3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "9e-8" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "pinstripes", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "pinstripes/ps/qwen3-coder-30b-a3b", "name": "pinstripes/ps/qwen3-coder-30b-a3b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "pinstripes", "maxInputTokens": 131072, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": false, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "pinstripes/ps/deepseek-v4-flash", "name": "pinstripes/ps/deepseek-v4-flash", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.09999999999999999, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1e-7" } ], "output": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "pinstripes", "maxInputTokens": 163840, "maxOutputTokens": 163840, "maxTokens": 163840, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "pinstripes/ps/minimax-m2.7", "name": "pinstripes/ps/minimax-m2.7", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.255, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.55e-7" } ], "output": [ { "amount": 0.55, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "5.5e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "pinstripes", "maxInputTokens": 1000192, "maxOutputTokens": 1000192, "maxTokens": 1000192, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": false, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "darkbloom/gemma-4-26b", "name": "darkbloom/gemma-4-26b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.03, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3e-8" } ], "output": [ { "amount": 0.165, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.65e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "darkbloom", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "darkbloom/gpt-oss-20b", "name": "darkbloom/gpt-oss-20b", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.0145, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.45e-8" } ], "output": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7e-8" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "darkbloom", "maxInputTokens": 131072, "maxOutputTokens": 32768, "maxTokens": 32768, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": true, "webSearch": null } } }, { "id": "xai/grok-4.20-0309-non-reasoning", "name": "xai/grok-4.20-0309-non-reasoning", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-4.20-multi-agent-0309", "name": "xai/grok-4.20-multi-agent-0309", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1.25, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00000125" }, { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" } ], "output": [ { "amount": 2.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000025" }, { "amount": 5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000005" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 1000000, "maxOutputTokens": 1000000, "maxTokens": 1000000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": true } } }, { "id": "xai/grok-build-0.1", "name": "xai/grok-build-0.1", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" } ], "output": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000004" } ], "cacheRead": [ { "amount": 0.19999999999999998, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2e-7" }, { "amount": 0.39999999999999997, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4e-7" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "xai", "maxInputTokens": 256000, "maxOutputTokens": 256000, "maxTokens": 256000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-transcribe", "name": "gpt-transcribe", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.000075, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.000075" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-live-transcribe", "name": "gpt-live-transcribe", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0002833333333333333, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0002833333333333333" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "openai", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "gpt-realtime-translate", "name": "gpt-realtime-translate", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0005666666666666667, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0005666666666666667" } ] } ], "metadata": { "source": "litellm", "mode": "realtime", "servingProvider": "openai", "maxInputTokens": 16000, "maxOutputTokens": 2000, "maxTokens": 2000, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": true, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "claude-mythos-5", "name": "claude-mythos-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "claude-mythos-preview", "name": "claude-mythos-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "output": [ { "amount": 50, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00005" } ], "cacheRead": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000001" } ], "cacheWrite": [ { "amount": 12.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.0000125" }, { "amount": 20, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00002" } ], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "anthropic", "maxInputTokens": 1000000, "maxOutputTokens": 128000, "maxTokens": 128000, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": null, "audioOutput": null, "promptCaching": true, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "gemini/gemini-robotics-er-2-streaming-preview", "name": "gemini/gemini-robotics-er-2-streaming-preview", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.000002" }, { "amount": 2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "audio", "raw": "0.000002" } ], "output": [ { "amount": 10, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.00001" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "gemini", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": true, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": true } } }, { "id": "mistral/mistral-small-2603", "name": "mistral/mistral-small-2603", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0.15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.5e-7" } ], "output": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6e-7" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 262144, "maxTokens": 262144, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": true, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/labs-leanstral-1-5", "name": "mistral/labs-leanstral-1-5", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "chat", "servingProvider": "mistral", "maxInputTokens": 262144, "maxOutputTokens": 131072, "maxTokens": 131072, "deprecationDate": null, "supports": { "functionCalling": true, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": true, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/mistral-moderation-2603", "name": "mistral/mistral-moderation-2603", "provider": "litellm", "tiers": [ { "name": "default", "input": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "output": [ { "amount": 0, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "litellm", "mode": "moderation", "servingProvider": "mistral", "maxInputTokens": 131072, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": null, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/voxtral-mini-2602", "name": "mistral/voxtral-mini-2602", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.00005, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.00005" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "mistral", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } }, { "id": "mistral/voxtral-mini-transcribe-realtime-2602", "name": "mistral/voxtral-mini-transcribe-realtime-2602", "provider": "litellm", "tiers": [ { "name": "default", "input": [], "output": [], "cacheRead": [], "cacheWrite": [], "other": [ { "amount": 0.0001, "currency": "USD", "units": 1, "pricingType": "second", "raw": "0.0001" } ] } ], "metadata": { "source": "litellm", "mode": "audio_transcription", "servingProvider": "mistral", "maxInputTokens": null, "maxOutputTokens": null, "maxTokens": null, "deprecationDate": null, "supports": { "functionCalling": null, "parallelFunctionCalling": null, "vision": null, "audioInput": true, "audioOutput": null, "promptCaching": null, "reasoning": null, "responseSchema": null, "systemMessages": null, "webSearch": null } } } ] }, { "provider": "baseten", "source": { "url": "https://www.baseten.co/pricing/", "fetchedAt": "2026-08-19T00:56:30.655Z" }, "models": [ { "id": "kimi-k3", "name": "Kimi K3", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "15" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.3" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": null, "availability": [ "dedicated", "shared" ], "publisher": "Moonshot AI", "deployLink": "https://www.baseten.co/talk-to-us/?model=kimi-K3", "tryModelApiLink": "https://app.baseten.co/model-apis/moonshotai/Kimi-K3" } }, { "id": "kimi-k26", "name": "Kimi K2.6", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.95" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.16" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": null, "availability": [ "dedicated", "shared" ], "publisher": "Moonshot AI", "deployLink": "https://www.baseten.co/talk-to-us/?model=kimi-K26", "tryModelApiLink": "https://app.baseten.co/model-apis/moonshotai/Kimi-K2.6" } }, { "id": "kimi-k27-code", "name": "Kimi K2.7 Code", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 0.95, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.95" } ], "output": [ { "amount": 4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4" } ], "cacheRead": [ { "amount": 0.16, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.16" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": null, "availability": [ "dedicated", "shared" ], "publisher": "Moonshot AI", "deployLink": "https://www.baseten.co/talk-to-us/?model=kimi-K27-code", "tryModelApiLink": "https://app.baseten.co/model-apis/moonshotai/Kimi-K2.7-Code" } }, { "id": "inkling-small", "name": "Inkling-Small", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.5" } ], "output": [ { "amount": 1.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.2" } ], "cacheRead": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.1" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": false, "availability": [ "dedicated", "shared" ], "publisher": "Thinkings Machines Lab", "deployLink": "https://www.baseten.co/talk-to-us/?Inkling-Small", "tryModelApiLink": "https://app.baseten.co/model-apis/inkling-small" } }, { "id": "inkling", "name": "Inkling", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1" } ], "output": [ { "amount": 4.05, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.05" } ], "cacheRead": [ { "amount": 0.17, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.17" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": false, "availability": [ "dedicated", "shared" ], "publisher": "Thinkings Machines Lab", "deployLink": "https://www.baseten.co/talk-to-us/?Inkling", "tryModelApiLink": "https://app.baseten.co/model-apis/inkling" } }, { "id": "glm-52", "name": "GLM-5.2", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 1.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.4" } ], "output": [ { "amount": 4.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "4.4" } ], "cacheRead": [ { "amount": 0.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.14" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": null, "availability": [ "shared", "dedicated" ], "publisher": "Z AI", "deployLink": "https://www.baseten.co/talk-to-us?model=glm-52", "tryModelApiLink": "https://app.baseten.co/model-apis/zai-org/GLM-5.2" } }, { "id": "glm-52-fast", "name": "GLM-5.2 Fast", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 2.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.1" } ], "output": [ { "amount": 6.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "6.6" } ], "cacheRead": [ { "amount": 0.21, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.21" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": false, "availability": [ "shared" ], "publisher": "Z AI", "deployLink": "https://app.baseten.co/model-apis/glm-5-2-Fast", "tryModelApiLink": "https://app.baseten.co/model-apis/zai-org/GLM-5.2-Fast" } }, { "id": "glm-4-7", "name": "GLM 4.7", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.6" } ], "output": [ { "amount": 2.2, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.2" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.12" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": null, "availability": [ "dedicated", "shared" ], "publisher": "Z AI", "deployLink": "https://www.baseten.co/talk-to-us?model=glm-4.7", "tryModelApiLink": "https://app.baseten.co/model-apis/glm-4-7" } }, { "id": "nvidia-nemotron-ultra", "name": "NVIDIA Nemotron 3 Ultra", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 0.6, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.6" } ], "output": [ { "amount": 2.4, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "2.4" } ], "cacheRead": [ { "amount": 0.12, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.12" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": null, "availability": [ "dedicated", "shared" ], "publisher": "NVIDIA", "deployLink": "https://www.baseten.co/talk-to-us?model=nemotronultra", "tryModelApiLink": "https://app.baseten.co/model-apis/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B" } }, { "id": "deepseek-v4-flash-0731", "name": "DeepSeek-V4-Flash-0731", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 0.13, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.13" } ], "output": [ { "amount": 0.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.26" } ], "cacheRead": [ { "amount": 0.028, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.028" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": false, "availability": [ "shared" ], "publisher": "DeepSeek", "deployLink": "https://www.baseten.co/talk-to-us/?model=deepseek-v4-flash-0731", "tryModelApiLink": "https://app.baseten.co/model-apis/deepseek-ai/DeepSeek-V4-Flash-0731" } }, { "id": "deepseek-v4", "name": "DeepSeek V4 Pro", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 1.74, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1.74" } ], "output": [ { "amount": 3.48, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "3.48" } ], "cacheRead": [ { "amount": 0.145, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.145" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": null, "availability": [ "dedicated", "shared" ], "publisher": "DeepSeek", "deployLink": "https://www.baseten.co/talk-to-us/?DeepSeekV4", "tryModelApiLink": "https://app.baseten.co/model-apis/deepseek-v4-pro" } }, { "id": "gpt-oss-120b", "name": "GPT OSS 120B", "provider": "baseten", "tiers": [ { "name": "standard", "input": [ { "amount": 0.1, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.1" } ], "output": [ { "amount": 0.5, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "0.5" } ], "cacheRead": [], "cacheWrite": [], "other": [] } ], "metadata": { "source": "baseten", "isClosedModel": null, "availability": [ "dedicated", "shared" ], "publisher": "OpenAI", "deployLink": "https://app.baseten.co/deploy/baseten/gpt-oss-120b-h100-throughput", "tryModelApiLink": "https://app.baseten.co/model-apis/openai/gpt-oss-120b" } } ] }, { "provider": "wafer", "source": { "url": "https://pass.wafer.ai/v1/models", "fetchedAt": "2026-08-19T00:56:30.963Z" }, "models": [ { "id": "GLM-5.2", "name": "GLM-5.2", "provider": "wafer", "tiers": [ { "name": "standard", "input": [ { "amount": 1.26, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "126 cents/1M" } ], "output": [ { "amount": 3.96, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "396 cents/1M" } ], "cacheRead": [ { "amount": 0.23, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "23 cents/1M" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "wafer", "description": "General Language Model 5.2 — high-quality bilingual (EN/ZH) generation with strong coding and reasoning capabilities.", "tier": "serverless_only", "contextLength": 1048576, "maxOutputTokens": null, "zdrSupported": true, "ownedBy": "wafer", "capabilities": { "vision": false, "tools": true, "reasoning": true, "zdr": { "same_capabilities": true, "supported": true }, "messages": { "tools": true, "vision": false, "reasoning": true, "streaming": true, "supported": true, "tool_streaming": true }, "responses": { "tools": true, "streaming": true, "supported": true, "text_format": [ "text", "json_object", "json_schema" ], "raw_json_schema_text": true }, "chat_completions": { "n": true, "regex": true, "tools": true, "grammar": true, "streaming": true, "supported": true, "json_object": true, "json_schema": true, "tool_streaming": true, "json_schema_refs": true, "tools_with_response_format": true } } } }, { "id": "Kimi-K3", "name": "Kimi-K3", "provider": "wafer", "tiers": [ { "name": "standard", "input": [ { "amount": 3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "300 cents/1M" } ], "output": [ { "amount": 15, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "1500 cents/1M" } ], "cacheRead": [ { "amount": 0.3, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "30 cents/1M" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "wafer", "description": "Kimi K3 sparse MoE model, self-hosted on the Wafer fleet. Available serverless with a 1M-token context window and strong coding/agentic performance.", "tier": "serverless_only", "contextLength": 1048576, "maxOutputTokens": null, "zdrSupported": true, "ownedBy": "wafer", "capabilities": { "vision": true, "tools": true, "reasoning": true, "zdr": { "same_capabilities": true, "supported": true }, "messages": { "tools": true, "vision": true, "reasoning": true, "streaming": true, "supported": true, "tool_streaming": true }, "responses": { "tools": true, "streaming": true, "supported": true, "text_format": [ "text", "json_object", "json_schema" ], "raw_json_schema_text": true }, "chat_completions": { "n": true, "regex": true, "tools": true, "grammar": true, "streaming": true, "supported": true, "json_object": true, "json_schema": true, "tool_streaming": true, "json_schema_refs": true, "tools_with_response_format": true } } } }, { "id": "Kimi-K2.6", "name": "Kimi-K2.6", "provider": "wafer", "tiers": [ { "name": "standard", "input": [ { "amount": 1.14, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "114 cents/1M" } ], "output": [ { "amount": 4.8, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "480 cents/1M" } ], "cacheRead": [ { "amount": 0.19, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "19 cents/1M" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "wafer", "description": "Kimi K2.6 sparse MoE model with a 262K context window. Available serverless and not included in standard Wafer Pass. Non-ZDR only: requests with `Wafer-ZDR: required` are rejected.", "tier": "serverless_only", "contextLength": 262144, "maxOutputTokens": null, "zdrSupported": false, "ownedBy": "wafer", "capabilities": { "vision": true, "tools": true, "reasoning": true, "zdr": { "disabled_reason": "self_hosted_backend_decommissioned", "supported": false }, "messages": { "tools": true, "vision": true, "reasoning": true, "streaming": true, "supported": true, "tool_streaming": true }, "responses": { "tools": true, "streaming": true, "supported": true, "text_format": [ "text", "json_object", "json_schema" ], "raw_json_schema_text": true }, "chat_completions": { "n": false, "regex": "partitioned", "tools": true, "grammar": false, "streaming": true, "supported": true, "json_object": true, "json_schema": true, "tool_streaming": true, "json_schema_refs": true, "tools_with_response_format": true } } } }, { "id": "DeepSeek-V4-Flash-0731-Fast", "name": "DeepSeek-V4-Flash-0731-Fast", "provider": "wafer", "tiers": [ { "name": "standard", "input": [ { "amount": 0.28, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "28 cents/1M" } ], "output": [ { "amount": 0.56, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "56 cents/1M" } ], "cacheRead": [ { "amount": 0.07, "currency": "USD", "units": 1000000, "pricingType": "token", "modality": "text", "raw": "7 cents/1M" } ], "cacheWrite": [], "other": [] } ], "metadata": { "source": "wafer", "description": "The same model served for high TPS.", "tier": "serverless_only", "contextLength": 1048576, "maxOutputTokens": null, "zdrSupported": true, "ownedBy": "wafer", "capabilities": { "vision": false, "tools": true, "reasoning": true, "zdr": { "same_capabilities": true, "supported": true }, "messages": { "tools": true, "vision": false, "reasoning": true, "streaming": true, "supported": true, "tool_streaming": true }, "responses": { "tools": true, "streaming": true, "supported": true, "text_format": [ "text", "json_object", "json_schema" ], "raw_json_schema_text": true }, "chat_completions": { "n": true, "regex": false, "tools": true, "grammar": false, "streaming": true, "supported": true, "json_object": true, "json_schema": true, "tool_streaming": true, "json_schema_refs": true, "tools_with_response_format": true } } } } ] } ] }