{ "version": 2, "license": "CC-BY-4.0", "models": [ { "provider": "ElevenLabs", "provider_url": "https://elevenlabs.io", "model_id": "eleven_flash_v2_5", "display_name": "Eleven Flash v2.5", "featured": true, "price_per_1m_chars_usd": "50.0", "voice_quality": "neural", "languages": "32+", "ssml_supported": false, "voice_cloning": true, "output_formats": [ "mp3_44100_128", "pcm_16000", "wav_44100", "opus_48000_128", "ulaw_8000", "alaw_8000" ], "time_to_first_byte_ms": 75, "last_verified": "2026-09-02", "last_changed_at": "2026-05-05", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://elevenlabs.io/pricing/api", "notes": "Current low-latency flagship; eleven_turbo_v2_5 is deprecated and replaced by Flash v2.5. Pay-as-you-go rate $0.05/1K chars. ~10K voices available. Plain text input only (no SSML)." }, { "provider": "ElevenLabs", "provider_url": "https://elevenlabs.io", "model_id": "eleven_multilingual_v2", "display_name": "Eleven Multilingual v2", "price_per_1m_chars_usd": "100.0", "voice_quality": "neural", "languages": "29+", "ssml_supported": false, "voice_cloning": true, "last_verified": "2026-09-02", "last_changed_at": "2026-05-19", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://elevenlabs.io/docs/overview/models", "notes": "High-quality professional model for audiobooks, video narration, and rich emotional expression. 29 languages, max 10,000 chars per request. Pay-as-you-go billed at 1 credit per character; Flash/Turbo v2.5 are billed at 0.5 credits/char (hence 2x the Flash $50/1M-chars rate). Higher latency than Flash; not recommended for real-time agents." }, { "provider": "ElevenLabs", "provider_url": "https://elevenlabs.io", "model_id": "eleven_v3", "display_name": "Eleven v3", "featured": true, "price_per_1m_chars_usd": "100.0", "voice_quality": "neural", "languages": "70+", "ssml_supported": false, "emotion_control_supported": true, "voice_cloning": true, "last_verified": "2026-09-02", "last_changed_at": "2026-05-19", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://elevenlabs.io/docs/overview/models", "notes": "Most expressive ElevenLabs TTS model (GA after alpha). 70+ languages, max 5,000 chars per request. Supports inline audio tags ([whispers], [sighs], [laughs], [happily]) for emotion/delivery control instead of SSML. Higher latency than Flash/Turbo v2.5 — ElevenLabs explicitly recommends v2.5 Flash/Turbo for real-time use. Pay-as-you-go billed at 1 credit/char (same multiplier as Multilingual v2). PVCs (professional voice clones) not yet fully optimized for v3." }, { "provider": "ElevenLabs", "provider_url": "https://elevenlabs.io", "model_id": "eleven_v3_conversational", "display_name": "Eleven v3 Conversational", "price_per_1m_chars_usd": "50.0", "voice_quality": "neural", "languages": "70+", "ssml_supported": false, "emotion_control_supported": true, "time_to_first_byte_ms": 280, "last_verified": "2026-09-02", "last_changed_at": "2026-09-02", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://elevenlabs.io/pricing/api", "notes": "New: low-latency realtime variant of Eleven v3, tuned for voice agents/conversational use (\"v3 Conversational\" on pricing page; model_id eleven_v3_conversational confirmed via elevenlabs.io/docs/models). 70+ languages, ~280ms latency, exposed via Text to Dialogue WebSocket. Supports audio tags for emotion/delivery control (no SSML, consistent with rest of ElevenLabs lineup). Billed at the Flash tier rate (0.5 credits/char, $0.05/1K chars) vs. eleven_v3's 1 credit/char. Character limit, exact launch date, and voice-cloning support not stated on either source as of verification; not marked featured pending a controller decision. Added 2026-09-02 refresh." }, { "provider": "ElevenLabs", "provider_url": "https://elevenlabs.io", "model_id": "eleven_turbo_v2_5", "display_name": "Eleven Turbo v2.5", "price_per_1m_chars_usd": "50.0", "voice_quality": "neural", "languages": "32+", "ssml_supported": false, "voice_cloning": true, "deprecated_at": "2026-05-19", "replaced_by_model_id": "eleven_flash_v2_5", "last_verified": "2026-09-02", "last_changed_at": "2026-05-19", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://elevenlabs.io/docs/overview/models", "notes": "Deprecated per ElevenLabs models page — outclassed by and replaced by eleven_flash_v2_5. Still callable but not recommended for new applications. No official sunset date published; deprecated_at reflects verification date. Pay-as-you-go billed at 0.5 credits/char (same as Flash v2.5)." }, { "provider": "ElevenLabs", "provider_url": "https://elevenlabs.io", "model_id": "eleven_flash_v2", "display_name": "Eleven Flash v2", "price_per_1m_chars_usd": "50.0", "voice_quality": "neural", "languages": ["English"], "ssml_supported": false, "voice_cloning": true, "time_to_first_byte_ms": 75, "last_verified": "2026-09-02", "last_changed_at": "2026-07-02", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://elevenlabs.io/docs/overview/models", "notes": "English-only predecessor to Flash v2.5, still listed as an active model. Same pay-as-you-go rate as Flash/Turbo family ($0.05/1K chars per pricing page, which groups Flash and Turbo variants together). ~75ms latency. Plain text input only (no SSML). Previously absent from this dataset; added on 2026-07-02 refresh after confirming it is still live via docs.elevenlabs.io/models." }, { "provider": "OpenAI", "provider_url": "https://openai.com", "model_id": "tts-1", "display_name": "TTS-1", "price_per_1m_chars_usd": "15.0", "voice_quality": "neural", "languages": ["en"], "ssml_supported": false, "voice_cloning": false, "last_verified": "2026-09-02", "last_changed_at": "2026-05-05", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://developers.openai.com/api/docs/models/tts-1", "notes": "Standard quality. Price unchanged at $15/1M chars as of 2026-09-02 refresh; not listed on the deprecations page; not in the main model-catalog listing (audio section lists only gpt-4o-mini-tts) but confirmed active and priced via the model detail page and the pricing page (openai.com/docs/models and openai.com/docs/pricing redirect to developers.openai.com)." }, { "provider": "OpenAI", "provider_url": "https://openai.com", "model_id": "tts-1-hd", "display_name": "TTS-1 HD", "price_per_1m_chars_usd": "30.0", "voice_quality": "neural", "languages": ["en"], "ssml_supported": false, "voice_cloning": false, "last_verified": "2026-09-02", "last_changed_at": "2026-05-05", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://developers.openai.com/api/docs/models/tts-1-hd", "notes": "High-definition tier — 2x the price of tts-1 for higher quality output. Price unchanged at $30/1M chars as of 2026-09-02 refresh; not listed on the deprecations page; not in the main model-catalog listing (audio section lists only gpt-4o-mini-tts) but confirmed active and priced via the model detail page and the pricing page (openai.com/docs/models and openai.com/docs/pricing redirect to developers.openai.com)." }, { "provider": "OpenAI", "provider_url": "https://openai.com", "model_id": "gpt-4o-mini-tts", "display_name": "GPT-4o mini TTS", "featured": true, "price_per_1m_chars_usd": "20.0", "voice_quality": "neural", "languages": ["en"], "ssml_supported": false, "voice_cloning": false, "last_verified": "2026-09-02", "last_changed_at": "2026-05-19", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://developers.openai.com/api/docs/models/gpt-4o-mini-tts", "notes": "OpenAI's newer GPT-4o-based TTS. Native pricing is token-based — $0.60/1M text-input tokens + $12/1M audio-output tokens — not per character. OpenAI's published estimate is ~$0.015 per minute of audio; converted to ~$20/1M chars assuming ~150 WPM (~750 chars/min) for consistency with the Cartesia row. Actual $/1M chars varies with speech rate and language. Supports voice steering via natural-language instructions (style/emotion) instead of SSML. Latest snapshot gpt-4o-mini-tts-2025-12-15; unchanged as of 2026-09-02 refresh (gpt-4o-mini-tts-2025-03-20 remains available as a pinned snapshot alongside it, per the model detail page's snapshot list). Max input 2,000 tokens per request." }, { "provider": "Cartesia", "provider_url": "https://cartesia.ai", "model_id": "sonic-3.6", "display_name": "Sonic 3.6", "price_per_1m_chars_usd": "50.0", "voice_quality": "neural", "languages": "44+", "ssml_supported": true, "voice_cloning": true, "output_formats": [ "raw/pcm_f32le", "raw/pcm_s16le", "raw/pcm_mulaw", "raw/pcm_alaw", "wav", "mp3" ], "time_to_first_byte_ms": 90, "last_verified": "2026-09-02", "last_changed_at": "2026-09-02", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://cartesia.ai/pricing", "notes": "New Cartesia flagship, GA August 27, 2026, superseding sonic-3.5 as the current-recommended TTS model. Latest snapshot sonic-3.6-2026-08-27; continuously-updated alias sonic-3.6; beta alias sonic-preview. 44 languages / 61 locales, adding Odia (or) and Urdu (ur) with instant voice cloning support (per https://docs.cartesia.ai/build-with-cartesia/tts-models/latest). Ranked #1 on Artificial Analysis's Speech Arena leaderboards per https://www.cartesia.ai/launch (secondary source); built on state-space-model architecture. Sub-90ms TTFB per https://www.cartesia.ai/launch; carried the existing 90ms convention pending a model-specific published number. SSML speed/volume tags apply (\"available on sonic-3 and later\" per https://docs.cartesia.ai/build-with-cartesia/sonic-3/ssml-tags); model achieves natural expression without requiring SSML. Pricing unchanged from sibling Sonic models: ~1 credit/char on every TTS endpoint (https://docs.cartesia.ai/pricing); USD anchor Pro plan $5/mo for 100K credits => $50/1M chars; no per-model pricing differentiation published. output_formats/sample-rate options are platform-level (shared with other Sonic snapshots), not documented separately per model. Added 2026-09-02 refresh; two-source verified via primary (cartesia.ai/pricing) plus docs.cartesia.ai/build-with-cartesia/tts-models/latest and /older-models." }, { "provider": "Cartesia", "provider_url": "https://cartesia.ai", "model_id": "sonic-3.5", "display_name": "Sonic 3.5", "featured": true, "price_per_1m_chars_usd": "50.0", "voice_quality": "neural", "languages": "42+", "ssml_supported": true, "voice_cloning": true, "output_formats": [ "raw/pcm_f32le", "raw/pcm_s16le", "raw/pcm_mulaw", "raw/pcm_alaw", "wav", "mp3" ], "time_to_first_byte_ms": 90, "replaced_by_model_id": "sonic-3.6", "confidence": "medium", "last_verified": "2026-09-02", "last_changed_at": "2026-05-05", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://cartesia.ai/pricing", "notes": "Cartesia now documents TTS billing as ~1 credit per character on every TTS endpoint (https://docs.cartesia.ai/pricing); Pro Voice Clone output is ~1.5 credits/char. USD anchor: Pro plan $5/mo for 100K credits => $50/1M chars; higher tiers are cheaper per credit (Startup ~$39.2, Scale ~$37.4 per 1M chars). Effective rate matches the prior $1-per-25-min + ~150 WPM anchoring, so no structured price change; per-char billing removes the speech-rate assumption. Plan-page minutes math (750 credits/min) is consistent with 1 credit/char at ~150 WPM. IVC voice cloning included (no clone fee). 90ms TTFB (sub-90ms per docs). 42 languages; latest snapshot sonic-3.5-2026-05-04. SSML speed/volume tags confirmed available on sonic-3.5 per https://docs.cartesia.ai/build-with-cartesia/sonic-3/ssml-tags (re-checked 2026-07-02). Emotion tags remain beta. Re-verified 2026-07-19: still Cartesia's current-recommended model. 2026-07-27 refresh: pricing and model status unchanged via primary (cartesia.ai/pricing) and secondary (docs.cartesia.ai/build-with-cartesia/tts-models/older-models) sources. 2026-08-11 refresh: pricing (Pro $5/mo => 133 TTS minutes, consistent with 100K credits at 750 credits/min), snapshot sonic-3.5-2026-05-04, 42 languages, and sub-90ms TTFB confirmed unchanged via primary (cartesia.ai/pricing) and secondary (docs.cartesia.ai/build-with-cartesia/tts-models/latest) sources; no newer Sonic model announced. 2026-09-02 refresh: still stable and callable (no sunset date published), but superseded as Cartesia's top-recommended model by sonic-3.6 (added as new row this pass); pricing and specs unchanged. confidence set to medium to reflect the same per-tier credit-rate variance as sibling rows now that this is no longer the flagship." }, { "provider": "Cartesia", "provider_url": "https://cartesia.ai", "model_id": "sonic-3", "display_name": "Sonic 3", "price_per_1m_chars_usd": "50.0", "voice_quality": "neural", "languages": "44+", "ssml_supported": true, "voice_cloning": true, "replaced_by_model_id": "sonic-3.6", "confidence": "medium", "last_verified": "2026-09-02", "last_changed_at": "2026-07-02", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://docs.cartesia.ai/build-with-cartesia/tts-models/older-models", "notes": "Predecessor to sonic-3.5, still stable and a valid model_id on the TTS API. Latest snapshot sonic-3-2026-01-12; older-models page lists 44 languages for this snapshot, updated from prior 42+ count (non-price change). Cartesia recommends sonic-3.5 for best results, most languages, and naturalness. SSML speed/volume tags confirmed available per https://docs.cartesia.ai/build-with-cartesia/sonic-3/ssml-tags. TTFB not published for this specific snapshot. Pricing equality with sonic-3.5 now documented: https://docs.cartesia.ai/pricing applies ~1 credit/char to every TTS endpoint; USD anchor Pro plan $5 per 100K credits => $50/1M chars. Confidence medium because the USD-per-credit rate varies by plan tier (Scale ~$37.4/1M chars). Model sunsets October 20, 2026 per https://docs.cartesia.ai/build-with-cartesia/tts-models/older-models. 2026-07-27 refresh: language count corrected to 44 (non-price), pricing and sunset date confirmed via secondary source. 2026-08-11 refresh: snapshot, 44 languages, pricing, and October 20 2026 sunset date confirmed unchanged via secondary source. 2026-09-02 refresh: snapshot, languages, pricing, and October 20 2026 sunset date confirmed unchanged via secondary source; replaced_by_model_id updated to sonic-3.6, Cartesia's new top-recommended model added this pass." }, { "provider": "Cartesia", "provider_url": "https://cartesia.ai", "model_id": "sonic-2", "display_name": "Sonic 2", "price_per_1m_chars_usd": "50.0", "voice_quality": "neural", "languages": ["en", "fr", "de", "es", "pt", "zh", "ja", "ko"], "ssml_supported": false, "voice_cloning": true, "time_to_first_byte_ms": 90, "replaced_by_model_id": "sonic-3.6", "confidence": "medium", "last_verified": "2026-09-02", "last_changed_at": "2026-05-19", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://docs.cartesia.ai/build-with-cartesia/tts-models/older-models", "notes": "Predecessor to sonic-3.5; still stable and callable, but Cartesia recommends sonic-3.5 for new builds. Latest snapshot sonic-2-2025-06-11, supporting 8 core languages (en, fr, de, es, pt, zh, ja, ko). The 7 additional languages previously carrying a 2026-06-01 EOL are no longer listed as supported on this snapshot as of 2026-07-02; languages field reduced from 15+ accordingly. 90ms model latency. Higher-fidelity voice cloning capability. Pricing equality with sonic-3.5 now documented (~1 credit/char on every TTS endpoint per https://docs.cartesia.ai/pricing); USD anchor Pro plan $5 per 100K credits => $50/1M chars. Confidence medium because the USD-per-credit rate varies by plan tier. Model sunsets October 20, 2026 per https://docs.cartesia.ai/build-with-cartesia/tts-models/older-models. 2026-07-27 refresh: pricing, snapshot, and language count confirmed unchanged; sunset date noted. 2026-08-11 refresh: pricing, snapshot, and language count confirmed unchanged via secondary source. 2026-09-02 refresh: pricing, snapshot, language count, and October 20 2026 sunset date confirmed unchanged via secondary source; replaced_by_model_id updated to sonic-3.6, Cartesia's new top-recommended model added this pass." }, { "provider": "Cartesia", "provider_url": "https://cartesia.ai", "model_id": "sonic-turbo", "display_name": "Sonic Turbo", "price_per_1m_chars_usd": "50.0", "voice_quality": "neural", "languages": ["en", "fr", "de", "es", "pt", "zh", "ja", "hi", "ko"], "ssml_supported": false, "voice_cloning": true, "time_to_first_byte_ms": 40, "replaced_by_model_id": "sonic-3.6", "confidence": "medium", "last_verified": "2026-09-02", "last_changed_at": "2026-05-19", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://docs.cartesia.ai/build-with-cartesia/tts-models/older-models", "notes": "Lowest-latency Sonic variant (~40ms TTFB). Still stable and callable, but Cartesia recommends sonic-3.5 for new builds. Latest snapshot sonic-turbo-2025-06-04, supporting 9 languages (en, fr, de, es, pt, zh, ja, hi, ko). The 6 additional languages previously carrying a 2026-06-01 EOL are no longer listed as supported on this snapshot as of 2026-07-02; languages field reduced from 15+ accordingly. Pricing equality with sonic-3.5 now documented (~1 credit/char on every TTS endpoint per https://docs.cartesia.ai/pricing); USD anchor Pro plan $5 per 100K credits => $50/1M chars. Confidence medium because the USD-per-credit rate varies by plan tier. Model sunsets October 20, 2026 per https://docs.cartesia.ai/build-with-cartesia/tts-models/older-models. 2026-07-27 refresh: pricing, snapshot, and language count confirmed unchanged; sunset date noted. 2026-08-11 refresh: pricing, snapshot, and language count confirmed unchanged via secondary source. 2026-09-02 refresh: pricing, snapshot, language count, and October 20 2026 sunset date confirmed unchanged via secondary source; replaced_by_model_id updated to sonic-3.6, Cartesia's new top-recommended model added this pass." }, { "provider": "Groq", "provider_url": "https://groq.com", "model_id": "canopy-labs-orpheus-english", "display_name": "Canopy Labs Orpheus English (Groq)", "price_per_1m_chars_usd": "22.0", "voice_quality": "neural", "languages": ["en"], "last_verified": "2026-09-02", "last_changed_at": "2026-05-05", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://console.groq.com/docs/models", "notes": "Hosted on Groq. Output speed ~100 characters/second. Vendor API model_id is canopylabs/orpheus-v1-english per console.groq.com/docs/models; this row keeps the pre-existing Hail slug for stability. Re-verified 2026-07-19 against groq.com/pricing and console.groq.com/docs/models: price and model_id unchanged; still listed as a preview model. 2026-08-11 refresh: groq.com/pricing now 308-redirects to the marketing homepage (no pricing content) and groq.com/groqcloud-models 404s; verified instead via console.groq.com/docs/models (price $22.00/1M chars, model_id, preview status unchanged) and console.groq.com/docs/deprecations (no deprecation scheduled for this model; playai-tts, its pre-Orpheus predecessor, was already shut down 2025-12-31). 2026-09-02 refresh: verified via console.groq.com/docs/models (price $22.00/1M chars, model_id, preview status unchanged) and console.groq.com/docs/deprecations (no new deprecation scheduled). No new Orpheus language variants found — only English and Arabic Saudi listed." }, { "provider": "Groq", "provider_url": "https://groq.com", "model_id": "canopy-labs-orpheus-arabic-saudi", "display_name": "Canopy Labs Orpheus Arabic Saudi (Groq)", "price_per_1m_chars_usd": "40.0", "voice_quality": "neural", "languages": ["ar-SA"], "last_verified": "2026-09-02", "last_changed_at": "2026-05-05", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://console.groq.com/docs/models", "notes": "Hosted on Groq. Saudi Arabic variant. Output speed ~100 characters/second. Vendor API model_id is canopylabs/orpheus-arabic-saudi per console.groq.com/docs/models; this row keeps the pre-existing Hail slug for stability. Re-verified 2026-07-19 against groq.com/pricing and console.groq.com/docs/models: price and model_id unchanged; still listed as a preview model. 2026-08-11 refresh: groq.com/pricing now 308-redirects to the marketing homepage (no pricing content) and groq.com/groqcloud-models 404s; verified instead via console.groq.com/docs/models (price $40.00/1M chars, model_id, preview status unchanged) and console.groq.com/docs/deprecations (no deprecation scheduled for this model; playai-tts-arabic, its pre-Orpheus predecessor, was already shut down 2025-12-31). 2026-09-02 refresh: verified via console.groq.com/docs/models (price $40.00/1M chars, model_id, preview status unchanged) and console.groq.com/docs/deprecations (no new deprecation scheduled). No new Orpheus language variants found." }, { "provider": "Google", "provider_url": "https://cloud.google.com", "model_id": "google-tts-studio", "display_name": "Google Cloud TTS — Studio", "price_per_1m_chars_usd": "160.0", "voice_quality": "neural", "languages": "40+", "ssml_supported": true, "voice_cloning": false, "deployment_options": ["native", "vertex"], "last_verified": "2026-09-02", "last_changed_at": "2026-05-19", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://cloud.google.com/text-to-speech/pricing", "notes": "Google's premium TTS tier for professional media production (long-form narration, advertising). Vendor price $0.00016/char = $160/1M chars (sku 84AB-48C0-F9C3); the single-speaker Studio class is GA and the multispeaker class is experimental per https://docs.cloud.google.com/text-to-speech/docs/voices. SSML supported except , , , and . Model_id is a Hail-coined tier slug (Google bills per-voice-tier rather than per API model name). Free tier: first 1M chars/month included per current pricing table (earlier rows recorded 100K; free tier is not representable in schema free_tier fields, and free-tier size is not a structured price field, so last_changed_at is not bumped). 2026-07-19 refresh: clean primary-source read achieved — raw curl of the pricing page HTML yielded the full pricing table (WebFetch still truncates). Studio is now grouped under the page's 'Legacy TTS models' section (no deprecation notice; still GA per secondary docs/voices). Price unchanged; confidence raised to high (primary table + secondary docs). 2026-08-11 refresh: price unchanged ($160/1M, sku 84AB-48C0-F9C3, still under 'Legacy TTS models'); no deprecation banner; secondary docs/voices corroborates GA status. 2026-09-02 refresh: price unchanged ($160/1M, sku 84AB-48C0-F9C3, still under 'Legacy TTS models'); no deprecation banner; secondary docs/voices (single-speaker Studio GA, multispeaker experimental) corroborates." }, { "provider": "Google", "provider_url": "https://cloud.google.com", "model_id": "google-tts-neural2", "display_name": "Google Cloud TTS — Neural2", "price_per_1m_chars_usd": "16.0", "voice_quality": "neural", "languages": "40+", "ssml_supported": true, "voice_cloning": false, "deployment_options": ["native", "vertex"], "last_verified": "2026-09-02", "last_changed_at": "2026-05-19", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://cloud.google.com/text-to-speech/pricing", "notes": "Google's recommended general-purpose neural tier (newer architecture than WaveNet; no longer the same rate — WaveNet dropped to the $4/1M Standard rate). Vendor price $0.000016/char = $16/1M chars (sku FEBD-04B6-769B, shared with the Polyglot preview tier). SSML fully supported. Model_id is a Hail-coined tier slug (Google bills per-voice-tier rather than per API model name). Free tier: first 1M chars/month included (not representable in schema free_tier fields). 2026-07-19 refresh: clean primary-source read achieved — raw curl of the pricing page HTML yielded the full pricing table (WebFetch still truncates). Neural2 now sits under the page's 'Legacy TTS models' section (no deprecation notice; still GA per secondary docs/voices). Price unchanged; confidence raised to high (primary table + secondary docs). 2026-08-11 refresh: price unchanged ($16/1M, sku FEBD-04B6-769B, still under 'Legacy TTS models'); no deprecation banner; secondary docs/voices corroborates GA status. 2026-09-02 refresh: price unchanged ($16/1M, sku FEBD-04B6-769B, still under 'Legacy TTS models'); no deprecation banner; secondary docs/voices corroborates GA status." }, { "provider": "Google", "provider_url": "https://cloud.google.com", "model_id": "google-tts-wavenet", "display_name": "Google Cloud TTS — WaveNet", "price_per_1m_chars_usd": "4.0", "voice_quality": "neural", "languages": "40+", "ssml_supported": true, "voice_cloning": false, "deployment_options": ["native", "vertex"], "last_verified": "2026-09-02", "last_changed_at": "2026-07-19", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://cloud.google.com/text-to-speech/pricing", "notes": "Original neural-net voice family from DeepMind. PRICE CHANGE 2026-07-19: $16/1M -> $4/1M chars. A clean primary-source read (raw curl of the pricing page HTML; WebFetch still truncates) shows WaveNet now listed under 'Legacy TTS models' and billed on the same SKU as Standard voices (sku 9D01-5995-B545) at $0.000004/char = $4/1M chars, with the free tier enlarged to the first 4M chars/month (was 1M). Cross-confirmed by third-party aggregators (costbench.com, diyai.io, xpay.sh) all listing WaveNet at $4/1M as of 2026-07; the WaveNet doc page (docs.cloud.google.com/text-to-speech/docs/wavenet) no longer states a price and defers to the pricing page. This retroactively validates the 2026-07-02 aggregator claim of $4/1M that the 07-13 pass dismissed as a tier mismatch. Not deprecated — no deprecation banner; still GA per secondary docs/voices — but Google recommends Neural2 ($16/1M) or Chirp 3: HD for new projects. SSML fully supported. Model_id is a Hail-coined tier slug. 2026-08-11 refresh: price unchanged ($4/1M, sku 9D01-5995-B545, 4M-char free tier, still under 'Legacy TTS models'); no deprecation banner; secondary docs/voices corroborates GA status. 2026-09-02 refresh: price unchanged ($4/1M, sku 9D01-5995-B545, 4M-char free tier, still under 'Legacy TTS models'); no deprecation banner; secondary docs/voices corroborates GA status." }, { "provider": "Google", "provider_url": "https://cloud.google.com", "model_id": "google-tts-chirp-3-hd", "display_name": "Google Cloud TTS — Chirp 3: HD", "price_per_1m_chars_usd": "30.0", "voice_quality": "neural", "languages": "30+", "ssml_supported": false, "streaming_supported": true, "voice_cloning": false, "deployment_options": ["native", "vertex"], "last_verified": "2026-09-02", "last_changed_at": "2026-05-19", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://cloud.google.com/text-to-speech/pricing", "notes": "Google's newest-generation TTS family with 30 voice styles in 30+ languages. Vendor price $0.00003/char = $30/1M chars (sku F977-2280-6F1B), listed under the pricing page's 'Latest TTS models' section. Per https://docs.cloud.google.com/text-to-speech/docs/chirp3-hd, Chirp 3: HD explicitly does NOT support SSML, speaking-rate adjustments, or pitch parameters; streaming synthesis IS supported. Model_id is a Hail-coined tier slug. Free tier: first 1M chars/month included. 2026-07-19 refresh: clean primary-source read achieved — raw curl of the pricing page HTML yielded the full pricing table (WebFetch still truncates); tier still GA with unchanged SSML/streaming behavior per secondary docs/voices. Price unchanged; confidence raised to high (primary table + secondary docs). 2026-08-11 refresh: price unchanged ($30/1M, sku F977-2280-6F1B, still under 'Latest TTS models'); no deprecation banner; secondary docs/voices corroborates GA status and no-SSML behavior. 2026-09-02 refresh: price unchanged ($30/1M, sku F977-2280-6F1B, still under 'Latest TTS models'); no deprecation banner; secondary docs/voices corroborates GA status and no-SSML behavior." }, { "provider": "Google", "provider_url": "https://cloud.google.com", "model_id": "google-tts-instant-custom-voice", "display_name": "Google Cloud TTS — Chirp 3: Instant custom voice", "price_per_1m_chars_usd": "60.0", "voice_quality": "cloned", "languages": "30+", "streaming_supported": true, "voice_cloning": true, "deployment_options": ["native", "vertex"], "confidence": "medium", "last_verified": "2026-09-02", "last_changed_at": "2026-07-19", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://cloud.google.com/text-to-speech/pricing", "notes": "Row added 2026-07-19. Chirp 3: Instant custom voice — voice cloning from ~10s of reference audio plus a recorded consent statement, synthesized via a reusable voice cloning key. Vendor price $0.00006/char = $60/1M chars (sku A247-37D7-C094), listed under the pricing page's 'Latest TTS models' section; no free tier ('Not available' on the pricing table). Sources: primary pricing page (full table via raw curl of the page HTML) + https://docs.cloud.google.com/text-to-speech/docs/chirp3-instant-custom-voice. 34 languages; streaming and long-form synthesis supported; supports Chirp 3 pace controls and pause tags (SSML support not documented, so ssml_supported omitted). Model_id is a Hail-coined tier slug. confidence medium: the feature is in preview and access is restricted to allow-listed users, though the price itself is published on the primary pricing table. 2026-08-11 refresh: price unchanged ($60/1M, sku A247-37D7-C094, no free tier, still under 'Latest TTS models'); still preview/allow-listed, confidence remains medium; secondary docs/voices does not list this row (stale relative to the pricing page, not a contradiction). 2026-09-02 refresh: price unchanged ($60/1M, sku A247-37D7-C094, no free tier, still under 'Latest TTS models'); still preview/allow-listed, confidence remains medium; secondary docs/voices still does not list this row." }, { "provider": "Microsoft Azure", "provider_url": "https://azure.microsoft.com", "model_id": "azure-tts-neural", "display_name": "Azure AI Speech — Neural", "price_per_1m_chars_usd": "15.0", "voice_quality": "neural", "languages": "100+", "ssml_supported": true, "streaming_supported": true, "voice_cloning": false, "deployment_options": ["azure"], "confidence": "medium", "last_verified": "2026-09-02", "last_changed_at": "2026-07-13", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/speech-services/", "notes": "Azure's standard neural TTS tier (called 'Neural' on the pricing page; 'Standard voice' in docs). 500+ prebuilt voices across 100+ locales per https://learn.microsoft.com/en-us/azure/ai-services/speech-service/text-to-speech (reconfirmed 2026-07-13: 100+ languages/locales, full SSML support, unchanged). S0 pay-as-you-go price for both real-time and batch synthesis (HD, AOAI, Custom Neural Voice, and Personal Voice priced separately). Chinese characters counted as 2 chars for billing. Free tier (F0): 500K chars/month. 2026-07-13 refresh: primary marketing pricing page again unreachable via WebFetch (timeout across 4 attempts, multiple URL variants — same persistent issue noted in the 2026-07-02 refresh). Queried Microsoft's own public Azure Retail Prices API directly (https://prices.azure.com/api/retail/prices, meterName 'S1 Neural Text To Speech Characters', productName 'Azure Speech', meterId 0f98e708-a16c-407b-8089-a0ed9e14ab49): retailPrice is $15.00/1M chars for the Global meter and all commercial regions (US Gov regions bill $18.75/1M), with effectiveStartDate 2024-02-01 — i.e. this rate has been continuously in effect since Feb 2024, not a fresh vendor price change this week. This is Microsoft's own first-party billing-meter API, more authoritative than the marketing page or any third-party recap. price_per_1m_chars_usd corrected from '16.0' to '15.0' to match the live meter — this appears to fix a stale figure carried forward from earlier refreshes. Note: multiple third-party aggregators (costbench.com, texttolab.com) and general web search still report '$16/1M' as of mid-2026, apparently echoing each other or a stale marketing-page snapshot rather than the live billing meter; none could be reconciled against prices.azure.com. confidence held at medium pending a clean fetch of the live marketing pricing page to confirm its displayed sticker price matches the billing meter. 2026-07-19 refresh: marketing pricing page still unreachable via WebFetch (ETIMEDOUT). $15.00/1M reconfirmed unchanged via Azure Retail Prices API (same meterId 0f98e708-a16c-407b-8089-a0ed9e14ab49, retailPrice 15.0 per 1M for the Global meter and all commercial regions, effectiveStartDate 2024-02-01 — queried 2026-07-19). 100+ languages/locales, full SSML support, and Chinese-characters-count-as-2 billing reconfirmed via learn.microsoft.com/en-us/azure/ai-services/speech-service/text-to-speech (doc now brands the service 'Azure Speech in Foundry Tools'; tier still called 'Neural' on the pricing page / 'Standard voice' in docs). No structured field changed. 2026-08-11 refresh: primary marketing pricing page still rendered client-side '$-' placeholders (WebFetch reached the page but prices are JS-populated, not a fetch failure). $15.00/1M reconfirmed unchanged via Azure Retail Prices API (same meterId 0f98e708-a16c-407b-8089-a0ed9e14ab49, Global meter, effectiveStartDate still 2024-02-01). 100+ languages/locales and full SSML support reconfirmed via learn.microsoft.com/en-us/azure/ai-services/speech-service/text-to-speech (page unchanged since the 2026-07-13 pass, still branded 'Azure Speech in Foundry Tools'). No structured field changed. 2026-09-02 refresh: primary marketing pricing page again rendered client-side '$-' placeholders (WebFetch reached the page, prices are JS-populated). $15.00/1M reconfirmed unchanged via Azure Retail Prices API (same meterId 0f98e708-a16c-407b-8089-a0ed9e14ab49, Global meter, effectiveStartDate still 2024-02-01, queried directly with curl since the field is only visible via the raw API, not the WebFetch-rendered page). 100+ languages/locales and SSML support reconfirmed unchanged via learn.microsoft.com/en-us/azure/ai-services/speech-service/text-to-speech (still branded 'Azure Speech in Foundry Tools'). No structured field changed." }, { "provider": "Microsoft Azure", "provider_url": "https://azure.microsoft.com", "model_id": "azure-tts-hd", "display_name": "Azure AI Speech — Neural HD (DragonHD)", "aliases": ["dragon-hd"], "price_per_1m_chars_usd": "22.0", "voice_quality": "neural", "voices_count": 30, "languages": ["en-US", "zh-CN", "de-DE", "es-ES", "fr-FR", "ja-JP"], "ssml_supported": false, "emotion_control_supported": true, "streaming_supported": true, "voice_cloning": false, "deployment_options": ["azure"], "confidence": "medium", "last_verified": "2026-09-02", "last_changed_at": "2026-03-01", "verification_method": "manual-confirmed", "verified_by": "r13i", "source_url": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/speech-services/", "notes": "Azure's premium HD neural tier (DragonHD architecture, 30+ GA voices). Per https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/azure-speech-%E2%80%93-neural-hd-text-to-speech-recent-voice-updates/4505380, Azure reduced Neural HD pricing to $22/1M chars effective March 2026 (down from $30/1M). Latency <300ms, real-time only. SSML support is partial; per the current supported/unsupported SSML table at learn.microsoft.com/en-us/azure/ai-services/speech-service/high-definition-voices, DragonHD (non-Omni) supports , , , (alias only), , , ,

, , but NOT , , ,