{ "openapi": "3.1.0", "info": { "title": "Gradium API", "description": "This documentation covers the Gradium API.\n\nThis API exposes our Text-To-Speech and Speech-To-Text models, which offers low-latency, high-quality & natural sounding output and best in class accuracy. \n\nFor issues, questions, or feature requests, please contact us at support@gradium.ai", "version": "0.1.0" }, "servers": [ { "url": "https://api.gradium.ai/api", "description": "Gradium API" } ], "paths": { "/voices/": { "post": { "tags": [ "Voices" ], "summary": "Create Voice", "description": "Create a new voice for an organization with audio file upload.", "operationId": "create_voice_voices__post", "requestBody": { "required": true, "content": { "multipart/form-data": { "schema": { "$ref": "#/components/schemas/Body_create_voice_voices__post" } } } }, "responses": { "201": { "description": "Successful Response", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/VoiceCreateResponse" } } } }, "422": { "description": "Validation Error", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/HTTPValidationError" } } } } } }, "get": { "tags": [ "Voices" ], "summary": "Get Voices", "description": "List voices for the authenticated organization.", "operationId": "get_voices_voices__get", "parameters": [ { "name": "skip", "in": "query", "required": false, "schema": { "type": "integer", "minimum": 0, "default": 0, "title": "Skip" } }, { "name": "limit", "in": "query", "required": false, "schema": { "type": "integer", "minimum": 0, "default": 100, "title": "Limit" } }, { "name": "include_catalog", "in": "query", "required": false, "schema": { "type": "boolean", "default": false, "title": "Include Catalog" } } ], "responses": { "200": { "description": "Successful Response", "content": { "application/json": { "schema": { "type": "array", "items": { "$ref": "#/components/schemas/APIVoiceResponse" }, "title": "Response Get Voices Voices Get" } } } }, "422": { "description": "Validation Error", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/HTTPValidationError" } } } } } } }, "/voices/{voice_uid}": { "get": { "tags": [ "Voices" ], "summary": "Get Voice", "description": "Get a voice by its UID. Optional org_uid and key_uid for access control.", "operationId": "get_voice_voices__voice_uid__get", "parameters": [ { "name": "voice_uid", "in": "path", "required": true, "schema": { "type": "string", "title": "Voice Uid" } } ], "responses": { "200": { "description": "Successful Response", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/APIVoiceResponse" } } } }, "422": { "description": "Validation Error", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/HTTPValidationError" } } } } } }, "put": { "tags": [ "Voices" ], "summary": "Update Voice", "description": "Update a voice by its UID.", "operationId": "update_voice_voices__voice_uid__put", "parameters": [ { "name": "voice_uid", "in": "path", "required": true, "schema": { "type": "string", "title": "Voice Uid" } } ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/VoiceUpdate" } } } }, "responses": { "200": { "description": "Successful Response", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/VoiceResponse" } } } }, "422": { "description": "Validation Error", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/HTTPValidationError" } } } } } }, "delete": { "tags": [ "Voices" ], "summary": "Delete Voice", "description": "Delete a voice by its UID.", "operationId": "delete_voice_voices__voice_uid__delete", "parameters": [ { "name": "voice_uid", "in": "path", "required": true, "schema": { "type": "string", "title": "Voice Uid" } } ], "responses": { "204": { "description": "Successful Response" }, "422": { "description": "Validation Error", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/HTTPValidationError" } } } } } } }, "/pronunciations/": { "post": { "tags": [ "Pronunciations" ], "summary": "Create Pronunciation Dictionary", "description": "Create a pronunciation dictionary for the authenticated organization.", "operationId": "create_pronunciation_dictionary_pronunciations__post", "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PronunciationDictionaryCreate" } } } }, "responses": { "201": { "description": "Successful Response", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PronunciationDictionaryResponse" } } } }, "422": { "description": "Validation Error", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/HTTPValidationError" } } } } } }, "get": { "tags": [ "Pronunciations" ], "summary": "List Pronunciation Dictionaries", "description": "List pronunciation dictionaries for the authenticated organization.", "operationId": "list_pronunciation_dictionaries_pronunciations__get", "parameters": [ { "name": "limit", "in": "query", "required": false, "schema": { "type": "integer", "maximum": 1000, "minimum": 0, "default": 100, "title": "Limit" } }, { "name": "offset", "in": "query", "required": false, "schema": { "type": "integer", "minimum": 0, "default": 0, "title": "Offset" } }, { "name": "language", "in": "query", "required": false, "schema": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Language" } } ], "responses": { "200": { "description": "Successful Response", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PronunciationDictionaryListResponse" } } } }, "422": { "description": "Validation Error", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/HTTPValidationError" } } } } } } }, "/pronunciations/{uid}": { "get": { "tags": [ "Pronunciations" ], "summary": "Get Pronunciation Dictionary", "description": "Get a pronunciation dictionary by its UID.", "operationId": "get_pronunciation_dictionary_pronunciations__uid__get", "parameters": [ { "name": "uid", "in": "path", "required": true, "schema": { "type": "string", "title": "Uid" } } ], "responses": { "200": { "description": "Successful Response", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PronunciationDictionaryResponse" } } } }, "422": { "description": "Validation Error", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/HTTPValidationError" } } } } } }, "put": { "tags": [ "Pronunciations" ], "summary": "Update Pronunciation Dictionary", "description": "Update a pronunciation dictionary by its UID.", "operationId": "update_pronunciation_dictionary_pronunciations__uid__put", "parameters": [ { "name": "uid", "in": "path", "required": true, "schema": { "type": "string", "title": "Uid" } } ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PronunciationDictionaryUpdate" } } } }, "responses": { "200": { "description": "Successful Response", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/PronunciationDictionaryResponse" } } } }, "422": { "description": "Validation Error", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/HTTPValidationError" } } } } } }, "delete": { "tags": [ "Pronunciations" ], "summary": "Delete Pronunciation Dictionary", "description": "Delete a pronunciation dictionary by its UID.", "operationId": "delete_pronunciation_dictionary_pronunciations__uid__delete", "parameters": [ { "name": "uid", "in": "path", "required": true, "schema": { "type": "string", "title": "Uid" } } ], "responses": { "204": { "description": "Successful Response" }, "422": { "description": "Validation Error", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/HTTPValidationError" } } } } } } }, "/usages/credits": { "get": { "tags": [ "metering" ], "summary": "Get Credits", "description": "Get current credit balance for the authenticated user's subscription.", "operationId": "get_credits_usages_credits_get", "responses": { "200": { "description": "Successful Response", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/CreditsSummary" } } } } } } }, "/speech/tts": { "get": { "tags": [ "TTS" ], "summary": "TTS WebSocket Stream", "description": "Connect to this endpoint via WebSocket for real-time text-to-speech conversion with low latency audio streaming.\n\n**Connection URL:**\n\n```\nwss://api.gradium.ai/api/speech/tts\n```\n\n**Authentication:**\nInclude your API key in the WebSocket connection header:\n- Header: `x-api-key: your_api_key`\n\n---\n\n## Quick Reference\n\n| Direction | Message Type | Example |\n|-----------|-------------|---------|\n| 🔵⬆️ Client→Server | Setup (first) | `{\"type\": \"setup\", \"voice_id\": \"YTpq7expH9539ERJ\", \"model_name\": \"default\", \"output_format\": \"wav\"}` |\n| 🟢⬇️ Server→Client | Ready | `{\"type\": \"ready\", \"request_id\": \"uuid\"}` |\n| 🔵⬆️ Client→Server | Text (stream) | `{\"type\": \"text\", \"text\": \"Hello, world!\"}` |\n| 🟢⬇️ Server→Client | Audio (stream) | `{\"type\": \"audio\", \"audio\": \"base64...\"}` |\n| 🟢⬇️ Server→Client | Text (stream) | `{\"type\": \"text\", \"text\": \"Hello\", \"start_s\": 0.2, \"stop_s\": 0.6}` |\n| 🔵⬆️ Client→Server | EndOfStream | `{\"type\": \"end_of_stream\"}` |\n| 🟢⬇️ Server→Client | AEndOfStream | `{\"type\": \"end_of_stream\"}` |\n| 🔴⬇️ Server→Client | Error | `{\"type\": \"error\", \"message\": \"Error description\", \"code\": 1008}` |\n\n---\n\n## Message Types\n\n### 1. Setup Message (First Message)\n\n**Direction:** Client → Server\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"setup\",\n \"model_name\": \"default\",\n \"voice_id\": \"YTpq7expH9539ERJ\",\n \"output_format\": \"wav\"\n}\n```\n\n**Fields:**\n- `type` (string, required): Must be \"setup\"\n- `model_name` (string, optional): The TTS model to use (default: \"default\")\n- `voice_id` (string, required): Voice ID from the library (e.g., \"YTpq7expH9539ERJ\" for Emma's voice) or custom voice ID\n- `output_format` (string, optional): Audio format (default: \"wav\"). One of \"wav\", \"pcm\", \"opus\", \"ulaw_8000\", \"mulaw_8000\", \"alaw_8000\", \"pcm_8000\", \"pcm_16000\", \"pcm_22050\", \"pcm_24000\", \"pcm_44100\", \"pcm_48000\".\n\n**Important:** This must be the very first message sent after connection. The server will close the connection if any other message is sent first.\n\n---\n\n### 2. Ready Message\n\n**Direction:** Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"ready\",\n \"request_id\": \"550e8400-e29b-41d4-a716-446655440000\"\n}\n```\n\n**Fields:**\n- `type` (string): Will be \"ready\"\n- `request_id` (string): Unique identifier for the session\n\nThis message is sent by the server after receiving the setup message, indicating that the connection is ready to receive text messages.\n\n---\n\n### 3. Text Message (Subsequent Messages)\n\n**Direction:** Client → Server\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"text\",\n \"text\": \"Hello, world!\"\n}\n```\n\n**Fields:**\n- `type` (string, required): Must be \"text\"\n- `text` (string, required): The text to be converted to speech\n\nSend text messages to be converted to speech. You can send multiple text messages in sequence. The server will stream audio back as it's generated.\n\n**Important: split on whitespace, not inside words or before punctuation.** When you send multiple text messages, the server inserts a single whitespace between the contents of consecutive messages. Sending `\"foo\"` followed by `\"bar\"` is therefore equivalent to sending `\"foo bar\"` (a whitespace is added between them), not `\"foobar\"`. Splitting a word across two messages will change its pronunciation. For the same reason, do not split trailing punctuation into its own message: sending `\"foo\"` followed by `\".\"` yields `\"foo .\"` rather than `\"foo.\"`. Keep each message aligned to a whitespace boundary, with any trailing punctuation attached to the preceding word.\n\n---\n\n### 4. Audio Response\n\n**Direction:** Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"audio\",\n \"audio\": \"base64_encoded_audio_data...\"\n}\n```\n\n**Fields:**\n- `type` (string): Will be \"audio\"\n- `audio` (string): Base64-encoded audio data in the requested format\n\nWhen using `\"pcm\"` output format, the audio will adhere to the following\nspecifications:\n- **Sample Rate**: 48000 Hz (48kHz)\n- **Format**: PCM (Pulse Code Modulation)\n- **Bit Depth**: 16-bit signed integer\n- **Channels**: Single channel (mono)\n- **Chunk Size**: 3840 samples per chunk (80ms at 48kHz)\n\nWhen using the `\"wav\"` output format, the audio chunks are in WAV format,\nat 48kHz, 16-bit signed integer mono.\n\nWhen using the `\"opus\"` output format, the audio chunks use the Opus codec\nwrapped in an Ogg container.\n\nAlternative output formats include `\"ulaw_8000\"`, `\"alaw_8000\"`, `\"pcm_8000\"`,\n`\"pcm_16000\"`, and `\"pcm_24000\"`.\n\n**Important:** Multiple audio messages will be streamed for each text message. Continue receiving until you detect the end of speech or receive a new message type.\n\n---\n\n### 5. Text Response\n\n**Direction:** Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"text\",\n \"text\": \"Hello\",\n \"start_s\": 0.2,\n \"stop_s\": 0.6\n}\n```\n\n**Fields:**\n- `type` (string): Will be \"text\"\n- `text` (string): The portion of text that has been generated into speech\n- `start_s` (float): Start time in seconds of this text segment in the audio\n- `stop_s` (float): Stop time in seconds of this text segment in the audio\n\nThe server sends text messages back to indicate which parts of the input text\nhave been processed into speech as well as the associated timestamps in the\naudio stream.\n\n---\n\n### 6. End Of Stream\n\n**Direction:** Client → Server and Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"end_of_stream\",\n}\n```\n\nThis message is sent by the client when it has submitted all the text that it\nwants to be considered. The server will then send back all the remaining audio\nuntil all the text has been processed, then an `EndOfStream` message, and then\ncloses the websocket connection.\n\n---\n\n## Error Handling\n\nWhen errors occur, the server sends an error message as JSON before closing the connection:\n\n**Error Message Format:**\n```json\n{\n \"type\": \"error\",\n \"message\": \"Error description explaining what went wrong\",\n \"code\": 1008\n}\n```\n\n**Common Error Codes:**\n- `1008`: Policy Violation (e.g., invalid API key, missing setup message)\n- `1011`: Internal Server Error (unexpected server-side error)\n\n---\n\n## Best Practices\n\n1. **Always send setup first**: The server expects a setup message immediately after connection\n2. **Handle audio streaming**: Audio responses are streamed in chunks - buffer and process appropriately\n3. **Implement reconnection logic**: Network issues happen - build in automatic reconnection with exponential backoff\n4. **Monitor connection health**: Implement ping/pong or periodic checks to detect stale connections\n5. **Graceful error handling**: Parse error messages and handle different error codes appropriately\n6. **Reuse connections**: For multiple utterances, keep the connection alive and send multiple text messages\n7. **Close cleanly**: Always close WebSocket connections properly when done\n\n---\n", "parameters": [ { "name": "x-api-key", "in": "header", "required": true, "schema": { "type": "string" }, "description": "Your Gradium API key" } ], "responses": { "101": { "description": "WebSocket connection established" } }, "x-codeSamples": [ { "lang": "cURL", "source": "wscat -c \"wss://api.gradium.ai/api/speech/tts\" \\\n -H \"x-api-key: your_api_key\"\n# After connection, paste:\n# {\"type\":\"setup\",\"voice_id\":\"YTpq7expH9539ERJ\",\"model_name\":\"default\",\"output_format\":\"wav\"}\n# {\"type\":\"text\",\"text\":\"Hello, world!\"}\n# {\"type\":\"end_of_stream\"}\n" }, { "lang": "Python", "source": "import asyncio\nimport base64\nimport json\n\nimport websockets\n\n\nasync def synthesise(api_key: str, voice_id: str, text: str) -> bytes:\n setup = {\n \"type\": \"setup\",\n \"voice_id\": voice_id,\n \"model_name\": \"default\",\n \"output_format\": \"wav\",\n }\n audio_chunks = []\n\n async with websockets.connect(\n \"wss://api.gradium.ai/api/speech/tts\",\n additional_headers={\"x-api-key\": api_key},\n ) as ws:\n await ws.send(json.dumps(setup))\n ready = json.loads(await ws.recv())\n assert ready[\"type\"] == \"ready\"\n\n await ws.send(json.dumps({\"type\": \"text\", \"text\": text}))\n await ws.send(json.dumps({\"type\": \"end_of_stream\"}))\n\n while True:\n msg = json.loads(await ws.recv())\n if msg[\"type\"] == \"audio\":\n audio_chunks.append(base64.b64decode(msg[\"audio\"]))\n elif msg[\"type\"] == \"end_of_stream\":\n break\n elif msg[\"type\"] == \"error\":\n raise RuntimeError(msg[\"message\"])\n\n return b\"\".join(audio_chunks)\n\n\naudio = asyncio.run(synthesise(\"your_api_key\", \"YTpq7expH9539ERJ\", \"Hello, world!\"))\nwith open(\"output.wav\", \"wb\") as f:\n f.write(audio)\n" } ] } }, "/speech/asr": { "get": { "tags": [ "STT" ], "summary": "STT WebSocket Stream", "description": "Connect to this endpoint via WebSocket for real-time speech-to-text conversion with streaming audio input.\n\n**Connection URL:**\n\n```\nwss://api.gradium.ai/api/speech/asr\n```\n\n**Authentication:**\nInclude your API key in the WebSocket connection header:\n- Header: `x-api-key: your_api_key`\n\n---\n\n## Quick Reference\n\n| Direction | Message Type | Example |\n|-----------|-------------|---------|\n| 🔵⬆️ Client→Server | Setup (first) | `{\"type\": \"setup\", \"model_name\": \"default\", \"input_format\": \"pcm\", \"json_config\": {\"language\": \"en\", \"delay_in_frames\": 16}}` |\n| 🟢⬇️ Server→Client | Ready | `{\"type\": \"ready\", \"request_id\": \"uuid\", \"model_name\": \"default\", \"sample_rate\": 24000}` |\n| 🔵⬆️ Client→Server | Audio | `{\"type\": \"audio\", \"audio\": \"base64...\"}` |\n| 🟢⬇️ Server→Client | Text (result) | `{\"type\": \"text\", \"text\": \"Hello world\", \"start_s\": 0.5}` |\n| 🟢⬇️ Server→Client | VAD (activity) | `{\"type\": \"step\", \"vad\": [...], \"step_idx\": 5, \"step_duration_s\": 0.08}` |\n| 🟢⬇️ Server→Client | End Text | `{\"type\": \"end_text\", \"stop_s\": 2.5}` |\n| 🔵⬆️ Client→Server | Flush | `{\"type\": \"flush\", \"flush_id\": 1}` |\n| 🟢⬇️ Server→Client | Flushed | `{\"type\": \"flushed\", \"flush_id\": 1}` |\n| 🔵⬆️ Client→Server | EndOfStream | `{\"type\": \"end_of_stream\"}` |\n| 🟢⬇️ Server→Client | EndOfStream | `{\"type\": \"end_of_stream\"}` |\n| 🔴⬇️ Server→Client | Error | `{\"type\": \"error\", \"message\": \"Error description\", \"code\": 1008}` |\n\n---\n\n## Message Types\n\n### 1. Setup Message (First Message)\n\n**Direction:** Client → Server\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"setup\",\n \"model_name\": \"default\",\n \"input_format\": \"pcm\",\n \"json_config\": {\n \"language\": \"en\",\n \"delay_in_frames\": 16\n }\n}\n```\n\n**Fields:**\n- `type` (string, required): Must be \"setup\"\n- `model_name` (string, optional): The Speech-To-Text model to use (default: \"default\")\n- `input_format` (string, optional): Audio format (default: \"wav\"). One of \"pcm\", \"pcm_8000\", \"pcm_16000\", \"pcm_22050\", \"pcm_24000\", \"pcm_44100\", \"pcm_48000\", \"wav\", \"opus\", \"ulaw_8000\", \"mulaw_8000\", \"alaw_8000\".\n- `json_config` (object or string, optional): Advanced STT settings, for example `{\"language\":\"en\",\"delay_in_frames\":16}`.\n\n**Important:** This must be the very first message sent after connection. The server will close the connection if any other message is sent first.\n\n---\n\n### 2. Ready Message\n\n**Direction:** Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"ready\",\n \"request_id\": \"550e8400-e29b-41d4-a716-446655440000\",\n \"model_name\": \"default\",\n \"sample_rate\": 24000,\n \"frame_size\": 1920,\n \"delay_in_frames\": 0,\n \"text_stream_names\": []\n}\n```\n\n**Fields:**\n- `type` (string): Will be \"ready\"\n- `request_id` (string): Unique identifier for the session\n- `model_name` (string): The Speech To Text model being used\n- `sample_rate` (integer): Expected sample rate in Hz (typically 24000)\n- `frame_size` (int): Number of samples by which the model processes data (typically 1920 which is equivalent to 80ms at 24kHz)\n- `delay_in_frames` (integer): Delay in audio frames for the model\n- `text_stream_names` (array): List of text stream names\n\nThis message is sent by the server after receiving the setup message, indicating that the connection is ready to receive audio.\n\n---\n\n### 3. Audio Message\n\n**Direction:** Client → Server\n**Format:** JSON Object (with binary audio data)\n\n```json\n{\n \"type\": \"audio\",\n \"audio\": \"base64_encoded_audio_data...\"\n}\n```\n\n**Fields:**\n- `type` (string, required): Must be \"audio\"\n- `audio` (string, required): Base64-encoded audio data\n\n**Audio Format Requirements (for PCM input):**\n- **Sample Rate**: 24000 Hz (24kHz)\n- **Format**: PCM (Pulse Code Modulation)\n- **Bit Depth**: 16-bit signed integer (little-endian)\n- **Channels**: Single channel (mono)\n- **Chunk Size**: Recommended 1920 samples per frame (80ms at 24kHz)\n\nWhen using `\"wav\"` input format, the audio must be a valid WAV file using\nPCM data (so `AudioFormat` = 1 in the WAV header). Supported bits per sample\nare 16, 24 and 32 bits.\n\nWhen using `\"opus\"` input format, the audio must be some ogg wrapped opus data\nstream.\n\nSend audio messages to be transcribed. You can send multiple audio messages in sequence. The server will stream text and VAD responses as it processes the audio.\n\n---\n\n### 4. Text Response\n\n**Direction:** Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"text\",\n \"text\": \"Hello world\",\n \"start_s\": 0.5,\n \"stream_id\": 0\n}\n```\n\n**Fields:**\n- `type` (string): Will be \"text\"\n- `text` (string): The transcribed text\n- `start_s` (float): Start time of the transcription in seconds\n- `stream_id` (integer or null): Stream identifier for tracking multiple concurrent streams\n\nText messages contain the transcribed speech. Multiple text messages will be streamed as the audio is processed.\n\n---\n\n### 5. VAD Response (Voice Activity Detection)\n\n**Direction:** Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"step\",\n \"vad\": [\n {\n \"horizon_s\": 0.5,\n \"inactivity_prob\": 0.05\n },\n {\n \"horizon_s\": 1.0,\n \"inactivity_prob\": 0.08\n },\n {\n \"horizon_s\": 2.0,\n \"inactivity_prob\": 0.12\n }\n ],\n \"step_idx\": 5,\n \"step_duration_s\": 0.08,\n \"total_duration_s\": 0.4\n}\n```\n\n**Fields:**\n- `type` (string): Will be \"step\"\n- `vad` (array): List of VAD predictions with future horizons\n - `horizon_s` (float): Lookahead duration in seconds\n - `inactivity_prob` (float): Probability that voice activity has ended by this horizon in seconds.\n- `step_idx` (integer): The step index (increments every 80ms)\n- `step_duration_s` (float): Duration of this step in seconds (typically 0.08)\n- `total_duration_s` (float): Total duration of audio processed so far\n\n**VAD Interpretation:**\n- VAD messages are emitted every 80ms (one per audio frame)\n- Use the `inactivity_prob` value from the longest horizon to determine if the speaker has likely finished\n- Higher `inactivity_prob` values indicate higher confidence that speaking has ended\n- Recommended threshold: Use `vad[2][\"inactivity_prob\"]` (third prediction) as the turn-taking indicator\n\n---\n\n### 6. End Text Response\n\n**Direction:** Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"end_text\",\n \"stop_s\": 2.5,\n \"stream_id\": 0\n}\n```\n\n**Fields:**\n- `type` (string): Will be \"end_text\"\n- `stop_s` (float): Stop time of last `text` message in seconds\n- `stream_id` (integer or null): Stream identifier\n\nSent when the previous text segment has a finished and its end timestamp is\navailable.\n\n---\n\n### 7. Flush Message\n\n**Direction:** Client → Server\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"flush\",\n \"flush_id\": 1\n}\n```\n\n**Fields:**\n- `type` (string, required): Must be \"flush\"\n- `flush_id` (integer, required): Identifier for this flush request, echoed back in the `flushed` reply.\n\nThis message can be sent by the client to request the server to flush any\nbuffered audio and return all outstanding text results immediately. The server\nwill respond with a `flushed` message containing the same `flush_id` once the\nflush is complete.\n\n### 8. End Of Stream\n\n**Direction:** Client → Server and Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"end_of_stream\"\n}\n```\n\nThis message is sent by the client when it has finished sending audio. The server will then process any remaining audio and send back all outstanding text results, VAD information, and then an `end_of_stream` message before closing the connection.\n\n---\n\n## Error Handling\n\nWhen errors occur, the server sends an error message as JSON before closing the connection:\n\n**Error Message Format:**\n```json\n{\n \"type\": \"error\",\n \"message\": \"Error description explaining what went wrong\",\n \"code\": 1008\n}\n```\n\n**Common Error Codes:**\n- `1008`: Policy Violation (e.g., invalid API key, missing setup message, invalid audio format)\n- `1011`: Internal Server Error (unexpected server-side error)\n\n---\n\n## Best Practices for STT\n\n1. **Always send setup first**: The server expects a setup message immediately after connection\n2. **Use correct audio format**: When using PCM, ensure audio is 24kHz PCM 16-bit mono\n3. **Send appropriately sized chunks**: 1920 samples (80ms) per message is recommended\n4. **Graceful shutdown**: Send `end_of_stream` when done to properly close the session\n5. **VAD Threshold**: Our VAD provides estimated probabilities that the speaker would be silent for a fixed number of seconds in the future. The thresholds to trigger the end-of-the-turn decisions might be application-dependent; as a starting point we recommend looking at the horizon of 2s and trigger when the inactivity_prob is above 0.5: `turn_ended = msg[\"vad\"][2][\"inactivity_prob\"] > 0.5`.\n5. **Acting on VAD**: Whenever you decide that the VAD probabilities warrant a decision to consider the turn ended, there is still up to `delay_in_frames` audio frames processed by the model. Instead of feeding silence from the speaker, the system can be made more reactive by flushing the remainder of the turn's transcript. For that, you can feed in `delay_in_frames` chunks of silence (vectors of zeros). If those are fed in faster than realtime, the API also has a possibility to process them faster, allowing a considerably more reactive turn-around.\n", "parameters": [ { "name": "x-api-key", "in": "header", "required": true, "schema": { "type": "string" }, "description": "Your Gradium API key" } ], "responses": { "101": { "description": "WebSocket connection established" } }, "x-codeSamples": [ { "lang": "cURL", "source": "wscat -c \"wss://api.gradium.ai/api/speech/asr\" \\\n -H \"x-api-key: your_api_key\"\n# After connection, paste:\n# {\"type\":\"setup\",\"model_name\":\"default\",\"input_format\":\"pcm\",\"json_config\":{\"language\":\"en\",\"delay_in_frames\":16}}\n" }, { "lang": "Python", "source": "import asyncio\nimport base64\nimport json\n\nimport websockets\n\nCHUNK_BYTES = 1920 * 2 # 80 ms at 24 kHz, 16-bit mono.\n\n\nasync def transcribe(api_key: str, pcm_audio: bytes):\n setup = {\n \"type\": \"setup\",\n \"model_name\": \"default\",\n \"input_format\": \"pcm\",\n \"json_config\": {\n \"language\": \"en\",\n \"delay_in_frames\": 16,\n },\n }\n\n async with websockets.connect(\n \"wss://api.gradium.ai/api/speech/asr\",\n additional_headers={\"x-api-key\": api_key},\n ) as ws:\n await ws.send(json.dumps(setup))\n ready = json.loads(await ws.recv())\n assert ready[\"type\"] == \"ready\"\n\n async def producer():\n for off in range(0, len(pcm_audio), CHUNK_BYTES):\n chunk = pcm_audio[off : off + CHUNK_BYTES]\n await ws.send(json.dumps({\n \"type\": \"audio\",\n \"audio\": base64.b64encode(chunk).decode(),\n }))\n await ws.send(json.dumps({\"type\": \"end_of_stream\"}))\n\n async def consumer():\n while True:\n msg = json.loads(await ws.recv())\n if msg[\"type\"] == \"text\":\n print(msg[\"text\"])\n elif msg[\"type\"] == \"end_of_stream\":\n return\n elif msg[\"type\"] == \"error\":\n raise RuntimeError(msg[\"message\"])\n\n await asyncio.gather(producer(), consumer())\n\n\nasyncio.run(transcribe(\"your_api_key\", open(\"input.pcm\", \"rb\").read()))\n" } ] } }, "/speech/s2s": { "get": { "tags": [ "S2S" ], "summary": "S2S WebSocket Stream", "description": "Connect to this endpoint via WebSocket for real-time speech-to-speech: incoming audio is transcribed, optionally translated, and re-synthesized into speech.\n\n**Connection URL:**\n\n```\nwss://api.gradium.ai/api/speech/s2s\n```\n\n**Authentication:**\nInclude your API key in the WebSocket connection header:\n- Header: `x-api-key: your_api_key`\n\n---\n\n## Quick Reference\n\n| Direction | Message Type | Example |\n|-----------|-------------|---------|\n| 🔵⬆️ Client→Server | Setup (first) | `{\"type\": \"setup\", \"model_name\": \"default\", \"input_format\": \"pcm\", \"output_format\": \"pcm\", \"voice_id\": \"YTpq7expH9539ERJ\"}` |\n| 🟢⬇️ Server→Client | Ready | `{\"type\": \"ready\", \"request_id\": \"uuid\", \"sample_rate\": 48000}` |\n| 🔵⬆️ Client→Server | Audio | `{\"type\": \"audio\", \"audio\": \"base64...\"}` |\n| 🟢⬇️ Server→Client | Text (stream) | `{\"type\": \"text\", \"text\": \"Hello world\", \"start_s\": 0.5, \"stop_s\": 1.2}` |\n| 🟢⬇️ Server→Client | Audio (stream) | `{\"type\": \"audio\", \"audio\": \"base64...\"}` |\n| 🔵⬆️ Client→Server | EndOfStream | `{\"type\": \"end_of_stream\"}` |\n| 🟢⬇️ Server→Client | EndOfStream | `{\"type\": \"end_of_stream\"}` |\n| 🔴⬇️ Server→Client | Error | `{\"type\": \"error\", \"message\": \"Error description\", \"code\": 1008}` |\n\n---\n\n## Message Types\n\n### 1. Setup Message (First Message)\n\n**Direction:** Client → Server\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"setup\",\n \"model_name\": \"default\",\n \"input_format\": \"pcm\",\n \"output_format\": \"pcm\",\n \"voice_id\": \"YTpq7expH9539ERJ\"\n}\n```\n\n**Fields:**\n- `type` (string, required): Must be \"setup\"\n- `model_name` (string, optional): The speech-to-speech model to use (default: \"default\")\n- `stt_model_name` (string, optional): The speech-to-text model used to transcribe the input\n- `tts_model_name` (string, optional): The text-to-speech model used to synthesize the output\n- `input_format` (string, optional): Input audio format (default: \"wav\"). One of \"pcm\", \"pcm_8000\", \"pcm_16000\", \"pcm_22050\", \"pcm_24000\", \"pcm_44100\", \"pcm_48000\", \"wav\", \"opus\", \"ulaw_8000\", \"mulaw_8000\", \"alaw_8000\".\n- `output_format` (string, optional): Output audio format (default: \"wav\"). One of \"wav\", \"pcm\", \"opus\", \"ulaw_8000\", \"mulaw_8000\", \"alaw_8000\", \"pcm_8000\", \"pcm_16000\", \"pcm_22050\", \"pcm_24000\", \"pcm_44100\", \"pcm_48000\".\n- `voice_id` (string, optional): Voice ID from the library used for the synthesized output\n- `json_config` (object or string, optional): Advanced options. Set `target_language` to translate the speech (e.g. `{\"target_language\": \"en\"}`); omit it to keep the original language.\n\n**Important:** This must be the very first message sent after connection. The server will close the connection if any other message is sent first.\n\n---\n\n### 2. Ready Message\n\n**Direction:** Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"ready\",\n \"request_id\": \"550e8400-e29b-41d4-a716-446655440000\",\n \"sample_rate\": 48000,\n \"frame_size\": 3840\n}\n```\n\n**Fields:**\n- `type` (string): Will be \"ready\"\n- `request_id` (string): Unique identifier for the session\n- `sample_rate` (integer): Output sample rate in Hz\n- `frame_size` (integer): Output frame size in samples\n\nThis message is sent by the server after receiving the setup message, indicating that the connection is ready to receive audio.\n\n---\n\n### 3. Audio Message\n\n**Direction:** Client → Server\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"audio\",\n \"audio\": \"base64_encoded_audio_data...\"\n}\n```\n\n**Fields:**\n- `type` (string, required): Must be \"audio\"\n- `audio` (string, required): Base64-encoded input audio chunk\n\n**Audio Format Requirements (for PCM input):**\n- **Sample Rate**: 24000 Hz (24kHz)\n- **Format**: PCM (Pulse Code Modulation)\n- **Bit Depth**: 16-bit signed integer (little-endian)\n- **Channels**: Single channel (mono)\n- **Chunk Size**: Recommended 1920 samples per frame (80ms at 24kHz)\n\nSend audio messages to be converted. The server will stream back text and synthesized audio as it processes the input.\n\n---\n\n### 4. Text Response\n\n**Direction:** Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"text\",\n \"text\": \"Hello world\",\n \"start_s\": 0.5,\n \"stop_s\": 1.2,\n \"stream_id\": 0\n}\n```\n\n**Fields:**\n- `type` (string): Will be \"text\"\n- `text` (string): The transcribed (and translated, if `target_language` is set) text segment\n- `start_s` (float): Start time of the segment in seconds\n- `stop_s` (float): Stop time of the segment in seconds\n- `stream_id` (integer or null): Stream identifier\n\n---\n\n### 5. Audio Response\n\n**Direction:** Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"audio\",\n \"audio\": \"base64_encoded_audio_data...\",\n \"start_s\": 0.0,\n \"stop_s\": 0.08,\n \"stream_id\": 0\n}\n```\n\n**Fields:**\n- `type` (string): Will be \"audio\"\n- `audio` (string): Base64-encoded output audio chunk in the requested format\n- `start_s` (float): Start time of the chunk in seconds\n- `stop_s` (float): Stop time of the chunk in seconds\n- `stream_id` (integer or null): Stream identifier\n\nWhen using `\"pcm\"` output format, the audio is 16-bit signed integer mono. The output sample rate is reported in the `ready` message.\n\n---\n\n### 6. End Of Stream\n\n**Direction:** Client → Server and Server → Client\n**Format:** JSON Object\n\n```json\n{\n \"type\": \"end_of_stream\"\n}\n```\n\nThe client sends this when it has finished sending audio. The server then returns any remaining text and audio, an `end_of_stream` message, and closes the connection.\n\n---\n\n## Error Handling\n\nWhen errors occur, the server sends an error message as JSON before closing the connection:\n\n```json\n{\n \"type\": \"error\",\n \"message\": \"Error description explaining what went wrong\",\n \"code\": 1008\n}\n```\n\n**Common Error Codes:**\n- `1008`: Policy Violation (e.g., invalid API key, missing setup message, invalid audio format)\n- `1011`: Internal Server Error (unexpected server-side error)\n", "parameters": [ { "name": "x-api-key", "in": "header", "required": true, "schema": { "type": "string" }, "description": "Your Gradium API key" } ], "responses": { "101": { "description": "WebSocket connection established" } }, "x-codeSamples": [ { "lang": "cURL", "source": "wscat -c \"wss://api.gradium.ai/api/speech/s2s\" \\\n -H \"x-api-key: your_api_key\"\n# After connection, paste:\n# {\"type\":\"setup\",\"model_name\":\"default\",\"input_format\":\"pcm\",\"output_format\":\"pcm\",\"voice_id\":\"YTpq7expH9539ERJ\",\"json_config\":{\"target_language\":\"en\"}}\n" }, { "lang": "Python", "source": "import asyncio\nimport base64\nimport json\n\nimport websockets\n\nIN_CHUNK_BYTES = 1920 * 2 # 80 ms at 24 kHz, 16-bit mono.\n\n\nasync def speech_to_speech(api_key: str, pcm_audio: bytes, voice_id: str) -> bytes:\n setup = {\n \"type\": \"setup\",\n \"model_name\": \"default\",\n \"input_format\": \"pcm\",\n \"output_format\": \"pcm\",\n \"voice_id\": voice_id,\n \"json_config\": {\"target_language\": \"en\"},\n }\n out_audio = []\n\n async with websockets.connect(\n \"wss://api.gradium.ai/api/speech/s2s\",\n additional_headers={\"x-api-key\": api_key},\n ) as ws:\n await ws.send(json.dumps(setup))\n ready = json.loads(await ws.recv())\n assert ready[\"type\"] == \"ready\"\n\n async def producer():\n for off in range(0, len(pcm_audio), IN_CHUNK_BYTES):\n chunk = pcm_audio[off : off + IN_CHUNK_BYTES]\n await ws.send(json.dumps({\n \"type\": \"audio\",\n \"audio\": base64.b64encode(chunk).decode(),\n }))\n await ws.send(json.dumps({\"type\": \"end_of_stream\"}))\n\n async def consumer():\n while True:\n msg = json.loads(await ws.recv())\n if msg[\"type\"] == \"text\":\n print(msg[\"text\"])\n elif msg[\"type\"] == \"audio\":\n out_audio.append(base64.b64decode(msg[\"audio\"]))\n elif msg[\"type\"] == \"end_of_stream\":\n return\n elif msg[\"type\"] == \"error\":\n raise RuntimeError(msg[\"message\"])\n\n await asyncio.gather(producer(), consumer())\n\n return b\"\".join(out_audio)\n\n\nasyncio.run(speech_to_speech(\"your_api_key\", open(\"input.pcm\", \"rb\").read(), \"YTpq7expH9539ERJ\"))\n" } ] } }, "/post/speech/tts": { "post": { "tags": [ "TTS" ], "summary": "TTS POST Endpoint", "description": "Use this HTTP POST endpoint for simple, text-to-speech conversion. The audio\ndata is sent back in a streaming way.\n\n**Endpoint URL:**\n\n```\nhttps://api.gradium.ai/api/post/speech/tts\n```\n\n**Authentication:**\nInclude your API key in the request header:\n- Header: `x-api-key: your_api_key`\n\n---\n\n## Quick Example\n\n```bash\ncurl -L -X POST https://api.gradium.ai/api/post/speech/tts \\\n -H \"x-api-key: your_api_key\" \\\n -H \"Content-Type: application/json\" \\\n -d '{\"text\": \"Hello, this is a test of the text to speech system.\", \"voice_id\": \"YTpq7expH9539ERJ\", \"output_format\": \"wav\", \"only_audio\": true}' \\\n > output.wav\n```\n\n---\n\n## Request Format\n\n**Method:** POST\n**Content-Type:** application/json\n\n**Request Body:**\n```json\n{\n \"text\": \"Hello, this is a test of the text to speech system.\",\n \"voice_id\": \"YTpq7expH9539ERJ\",\n \"output_format\": \"wav\",\n \"json_config\": \"{}\",\n \"only_audio\": true\n}\n```\n\n**Fields:**\n- `text` (string, required): The text to be converted to speech\n- `voice_id` (string, required): Voice ID from the library (e.g.,\n \"YTpq7expH9539ERJ\") or a custom voice ID\n- `output_format` (string, required): Audio format - \"wav\", \"pcm\", or \"opus\"\n (ogg wrapped opus data).\n- `json_config` (string, optional): Additional configuration in JSON string format (e.g., `{\"padding_bonus\": -1.2}`)\n- `model_name` (string, optional): The TTS model to use (default: \"default\")\n- `only_audio` (boolean, optional): When `true`, returns only the raw audio\n bytes. When `false` or omitted, returns a stream of JSON messages containing\n the audio and metadata. The format is the same as with the websocket endpoint.\n\n---\n\n## Response Format\n\n### When `only_audio` is `true`\n\nThe response body contains the raw audio bytes in the requested format. Save directly to a file:\n\n```bash\ncurl ... > output.wav\n```\n\n**Content-Type:** Depends on the output format:\n- `audio/wav` for WAV format\n- `audio/ogg` for Ogg wrapped Opus format\n- `audio/pcm` for PCM format\n\n### When `only_audio` is `false` or omitted\n\nThe response is a stream of JSON messages using the same format as the\nWebSocket endpoint. Read the body line-by-line until it closes — the\nbody closing signals that synthesis is complete.\n\n## Error Handling\n\nIf the request fails before the response stream has started, the server\nresponds with `HTTP 500` and a plain-text body. Two body shapes occur:\n\n- **Upstream errors** (with a numeric code) such as authentication\n failures or worker-level rejections:\n\n ```\n error from server : \n ```\n\n For example, a revoked or expired API key returns\n `error from server 1008: API key is revoked or expired`.\n\n- **Proxy-level rejections** (e.g. unsupported `Content-Type`, malformed\n request body) come back as raw error strings without the `error from\n server` prefix.\n\nIn both cases the body is plain text (not JSON). Errors that occur\nafter the response stream has started (when `only_audio` is `false`)\nare surfaced as `{\"type\": \"error\", ...}` JSON messages within the\nstream rather than as a different HTTP status.\n\n---\n\n## When to Use POST vs WebSocket\n\nThe POST endpoint is ideal for simple, text-to-speech generations.\nThe main difference with the WebSocket endpoint is that the input is not\nhandled in a streaming way; the entire text is sent in one request. The audio is\nstill streamed back to the client, allowing for efficient handling of large\naudio outputs and lower latency.\n\nSo if your use case involves sending complete text blocks and receiving audio\nresponses, the POST endpoint is a straightforward choice. For more interactive\nor real-time applications where text input is streamed, the WebSocket endpoint\nis more suitable.\n", "parameters": [ { "name": "x-api-key", "in": "header", "required": true, "schema": { "type": "string" }, "description": "Your Gradium API key" } ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "type": "object", "required": [ "text", "voice_id", "output_format" ], "properties": { "text": { "type": "string", "description": "The text to convert to speech" }, "voice_id": { "type": "string", "description": "Voice ID from the library or custom voice ID" }, "output_format": { "type": "string", "enum": [ "wav", "pcm", "opus", "ulaw_8000", "mulaw_8000", "alaw_8000", "pcm_8000", "pcm_16000", "pcm_22050", "pcm_24000", "pcm_44100", "pcm_48000" ], "description": "Audio output format" }, "only_audio": { "type": "boolean", "description": "When true, returns raw audio bytes instead of JSON" } } } } } }, "responses": { "200": { "description": "Audio data returned successfully" }, "500": { "description": "Pre-stream error. Body is plain text. Upstream errors (authentication, worker rejections) are formatted as `error from server : `; proxy-level rejections (e.g. malformed request body) come back as raw error strings." } }, "x-codeSamples": [ { "lang": "cURL", "source": "curl -L -X POST https://api.gradium.ai/api/post/speech/tts \\\n -H \"x-api-key: your_api_key\" \\\n -H \"Content-Type: application/json\" \\\n -d '{\"text\": \"Hello, world!\", \"voice_id\": \"YTpq7expH9539ERJ\", \"output_format\": \"wav\", \"only_audio\": true}' \\\n > output.wav\n" }, { "lang": "Python", "source": "import requests\n\nresp = requests.post(\n \"https://api.gradium.ai/api/post/speech/tts\",\n json={\n \"text\": \"Hello, world!\",\n \"voice_id\": \"YTpq7expH9539ERJ\",\n \"output_format\": \"wav\",\n \"only_audio\": True,\n },\n headers={\"x-api-key\": \"your_api_key\"},\n)\nresp.raise_for_status()\nwith open(\"output.wav\", \"wb\") as f:\n f.write(resp.content)\n" } ] } }, "/post/speech/asr": { "post": { "tags": [ "STT" ], "summary": "STT POST Endpoint", "description": "Use this HTTP POST endpoint for simple, one-shot speech-to-text\ntranscription. Send the entire audio payload in the request body and receive\na stream of newline-delimited JSON (NDJSON) messages with the transcription\nresults.\n\n**Endpoint URL:**\n\n```\nhttps://api.gradium.ai/api/post/speech/asr\n```\n\n**Authentication:**\nInclude your API key in the request header:\n- Header: `x-api-key: your_api_key`\n\n---\n\n## Quick Example\n\n```bash\ncurl -L -X POST https://api.gradium.ai/api/post/speech/asr \\\n -H \"x-api-key: your_api_key\" \\\n -H \"Content-Type: audio/wav\" \\\n --data-binary @input.wav\n```\n\nWith a language hint:\n\n```bash\ncurl -L -X POST \"https://api.gradium.ai/api/post/speech/asr?json_config=%7B%22language%22%3A%22en%22%7D\" \\\n -H \"x-api-key: your_api_key\" \\\n -H \"Content-Type: audio/wav\" \\\n --data-binary @input.wav\n```\n\n---\n\n## Request Format\n\n**Method:** POST\n**Body:** Raw audio bytes (the full file).\n\nThe input audio format is selected from the `Content-Type` header:\n\n| Content-Type | Audio Format |\n|--------------|--------------|\n| `audio/wav` (default if header is missing) | WAV (PCM data, 16/24/32-bit) |\n| `audio/pcm` | Raw PCM, 24 kHz, 16-bit signed little-endian, mono |\n| `audio/ogg` or `audio/opus` | Ogg-wrapped Opus |\n\n**Query Parameters:**\n- `model` (string, optional): The Speech-to-Text model to use (default: `default`).\n- `input_format` (string, optional): Override the input format detected from\n `Content-Type`. One of `wav`, `pcm`, `opus`.\n- `json_config` (string, optional): JSON-encoded model configuration. Common\n use case: pass a language hint, e.g. `{\"language\": \"en\"}`. The value should\n be URL-encoded when used as a query parameter.\n\n---\n\n## Response Format\n\n**Content-Type:** `application/x-ndjson`\n\nThe response body is a stream of newline-delimited JSON messages. Each line\nis a separate JSON object. Possible message types:\n\n### `text` — transcribed text segment\n\n```json\n{\"type\": \"text\", \"text\": \"Hello world\", \"start_s\": 0.5, \"stream_id\": 0}\n```\n\n- `text` (string): Transcribed text.\n- `start_s` (float): Start time of the segment in seconds.\n- `stream_id` (integer): Stream identifier when multiple text streams are in\n use (0 in single-stream transcription).\n\n### `end_text` — segment boundary\n\n```json\n{\"type\": \"end_text\", \"stop_s\": 2.5, \"stream_id\": 0}\n```\n\n- `stop_s` (float): End time of the previous `text` segment in seconds.\n- `stream_id` (integer): Stream identifier.\n\n### `error` — server-side error\n\n```json\n{\"type\": \"error\", \"message\": \"Error description\"}\n```\n\nIf the transcription pipeline fails, the server emits an `error` message and\nstops the stream.\n\n---\n\n## Reading the Stream\n\nThe response is streamed: read the body line-by-line and parse each line as\nJSON. The body closes when transcription is complete.\n\n```python\nimport json\nimport requests\n\nwith open(\"input.wav\", \"rb\") as f:\n audio = f.read()\n\nwith requests.post(\n \"https://api.gradium.ai/api/post/speech/asr\",\n data=audio,\n headers={\n \"x-api-key\": \"your_api_key\",\n \"Content-Type\": \"audio/wav\",\n },\n stream=True,\n) as resp:\n resp.raise_for_status()\n transcript = []\n for line in resp.iter_lines(decode_unicode=True):\n if not line:\n continue\n msg = json.loads(line)\n if msg[\"type\"] == \"text\":\n transcript.append(msg[\"text\"])\n elif msg[\"type\"] == \"error\":\n raise RuntimeError(msg[\"message\"])\nprint(\" \".join(transcript))\n```\n\n---\n\n## Error Handling\n\nIf the request fails before the response stream has started, the server\nresponds with `HTTP 500` and a plain-text body. Two body shapes occur:\n\n- **Upstream errors** (with a numeric code) such as authentication\n failures or worker-level rejections:\n\n ```\n error from server : \n ```\n\n For example, a revoked or expired API key returns\n `error from server 1008: API key is revoked or expired`.\n\n- **Proxy-level rejections** (e.g. unsupported `Content-Type`, malformed\n request body) come back as raw error strings without the `error from\n server` prefix:\n\n ```\n unsupported content type for SST 'audio/mpeg'\n ```\n\nIn both cases the body is plain text (not JSON). Errors that occur\nafter the NDJSON stream has started are surfaced as\n`{\"type\": \"error\", \"message\": \"...\"}` lines within the stream rather\nthan as a different HTTP status.\n\n---\n\n## When to Use POST vs WebSocket\n\nThe POST endpoint is ideal for one-shot transcription of complete audio\nfiles already on disk or in memory. The audio is uploaded in a single\nrequest, transcription runs, and the results are streamed back as NDJSON.\n\nUse the [WebSocket endpoint](/api-reference/endpoint/stt-websocket) instead\nwhen you need to:\n- Stream audio as it is being captured (microphone, telephony).\n- Receive partial transcripts and Voice Activity Detection (VAD) events in\n real time for turn-taking.\n- Send a `flush` message to force the model to emit buffered text on demand.\n", "parameters": [ { "name": "x-api-key", "in": "header", "required": true, "schema": { "type": "string" }, "description": "Your Gradium API key" }, { "name": "Content-Type", "in": "header", "required": false, "schema": { "type": "string", "enum": [ "audio/wav", "audio/pcm", "audio/ogg", "audio/opus" ] }, "description": "Format of the audio in the request body. Defaults to audio/wav when omitted." }, { "name": "model", "in": "query", "required": false, "schema": { "type": "string", "default": "default" }, "description": "Speech-to-Text model name." }, { "name": "input_format", "in": "query", "required": false, "schema": { "type": "string", "enum": [ "wav", "pcm", "opus" ] }, "description": "Overrides the audio format detected from Content-Type." }, { "name": "json_config", "in": "query", "required": false, "schema": { "type": "string" }, "description": "JSON-encoded model configuration. Example: {\"language\": \"en\"}" } ], "requestBody": { "required": true, "content": { "audio/wav": { "schema": { "type": "string", "format": "binary", "description": "WAV audio file." } }, "audio/pcm": { "schema": { "type": "string", "format": "binary", "description": "Raw PCM audio: 24 kHz, 16-bit signed little-endian, mono." } }, "audio/ogg": { "schema": { "type": "string", "format": "binary", "description": "Ogg-wrapped Opus audio." } } } }, "responses": { "200": { "description": "NDJSON stream of transcription messages.", "content": { "application/x-ndjson": { "schema": { "type": "string", "description": "Newline-delimited JSON messages: text, end_text, or error. The body closes when transcription is complete." } } } }, "500": { "description": "Pre-stream error. Body is plain text. Upstream errors (authentication, worker rejections) are formatted as `error from server : `; proxy-level rejections (e.g. unsupported Content-Type) come back as raw error strings." } }, "x-codeSamples": [ { "lang": "cURL", "source": "curl -L -X POST https://api.gradium.ai/api/post/speech/asr \\\n -H \"x-api-key: your_api_key\" \\\n -H \"Content-Type: audio/wav\" \\\n --data-binary @input.wav\n" }, { "lang": "Python", "source": "import json\n\nimport requests\n\nwith open(\"input.wav\", \"rb\") as f:\n audio = f.read()\n\nwith requests.post(\n \"https://api.gradium.ai/api/post/speech/asr\",\n data=audio,\n headers={\n \"x-api-key\": \"your_api_key\",\n \"Content-Type\": \"audio/wav\",\n },\n stream=True,\n) as resp:\n resp.raise_for_status()\n for line in resp.iter_lines(decode_unicode=True):\n if not line:\n continue\n msg = json.loads(line)\n if msg[\"type\"] == \"text\":\n print(msg[\"text\"])\n" } ] } } }, "components": { "schemas": { "APIVoiceResponse": { "properties": { "uid": { "type": "string", "title": "Uid" }, "name": { "type": "string", "title": "Name" }, "description": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Description" }, "filename": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Filename" }, "start_s": { "anyOf": [ { "type": "number" }, { "type": "null" } ], "title": "Start S" }, "is_catalog": { "type": "boolean", "title": "Is Catalog" }, "is_pro_clone": { "type": "boolean", "title": "Is Pro Clone" }, "language": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Language" }, "tags": { "items": { "$ref": "#/components/schemas/ExportedTag" }, "type": "array", "title": "Tags", "default": [] } }, "type": "object", "required": [ "uid", "name", "is_catalog", "is_pro_clone" ], "title": "APIVoiceResponse", "description": "The response sent to the user in the API for external user." }, "Body_create_voice_voices__post": { "properties": { "audio_file": { "type": "string", "format": "binary", "title": "Audio File" }, "name": { "type": "string", "title": "Name" }, "input_format": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Input Format", "description": "Audio format. If omitted, inferred from the audio_file extension." }, "description": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Description" }, "language": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Language" }, "start_s": { "type": "number", "title": "Start S", "default": 0 }, "timeout_s": { "type": "number", "title": "Timeout S", "default": 10 } }, "type": "object", "required": [ "audio_file", "name" ], "title": "Body_create_voice_voices__post" }, "CreditsSummary": { "properties": { "remaining_credits": { "type": "integer", "title": "Remaining Credits" }, "allocated_credits": { "type": "integer", "title": "Allocated Credits" }, "billing_period": { "type": "string", "title": "Billing Period" }, "next_rollover_date": { "anyOf": [ { "type": "string", "format": "date" }, { "type": "null" } ], "title": "Next Rollover Date" }, "plan_name": { "type": "string", "title": "Plan Name", "default": "" } }, "type": "object", "required": [ "remaining_credits", "allocated_credits", "billing_period" ], "title": "CreditsSummary", "description": "Summary of credits for current billing period." }, "ExportedTag": { "properties": { "category": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Category" }, "value": { "type": "string", "title": "Value" } }, "type": "object", "required": [ "value" ], "title": "ExportedTag", "description": "Tag exported in the public API.\n\nHidden is used for filtering but not serialized." }, "HTTPValidationError": { "properties": { "detail": { "items": { "$ref": "#/components/schemas/ValidationError" }, "type": "array", "title": "Detail" } }, "type": "object", "title": "HTTPValidationError" }, "PronunciationDictionaryCreate": { "properties": { "name": { "type": "string", "title": "Name" }, "description": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Description" }, "language": { "type": "string", "title": "Language" }, "rules": { "items": { "$ref": "#/components/schemas/PronunciationRuleCreate" }, "type": "array", "title": "Rules", "default": [] } }, "type": "object", "required": [ "name", "language" ], "title": "PronunciationDictionaryCreate", "description": "Pronunciation dictionary create schema." }, "PronunciationDictionaryListResponse": { "properties": { "dictionaries": { "items": { "$ref": "#/components/schemas/PronunciationDictionaryResponse" }, "type": "array", "title": "Dictionaries" }, "total": { "type": "integer", "title": "Total" } }, "type": "object", "required": [ "dictionaries", "total" ], "title": "PronunciationDictionaryListResponse", "description": "Pronunciation dictionary list response schema." }, "PronunciationDictionaryResponse": { "properties": { "name": { "type": "string", "title": "Name" }, "description": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Description" }, "language": { "type": "string", "title": "Language" }, "uid": { "type": "string", "title": "Uid" }, "org_uid": { "type": "string", "format": "uuid", "title": "Org Uid" }, "created_at": { "type": "string", "format": "date-time", "title": "Created At" }, "rules": { "items": { "$ref": "#/components/schemas/PronunciationRuleResponse" }, "type": "array", "title": "Rules", "default": [] } }, "type": "object", "required": [ "name", "language", "uid", "org_uid", "created_at" ], "title": "PronunciationDictionaryResponse", "description": "Pronunciation dictionary response schema." }, "PronunciationDictionaryUpdate": { "properties": { "name": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Name" }, "description": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Description" }, "language": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Language" }, "rules": { "anyOf": [ { "items": { "$ref": "#/components/schemas/PronunciationRuleCreate" }, "type": "array" }, { "type": "null" } ], "title": "Rules" } }, "type": "object", "title": "PronunciationDictionaryUpdate", "description": "Pronunciation dictionary update schema." }, "PronunciationRuleCreate": { "properties": { "original": { "type": "string", "title": "Original" }, "rewrite": { "type": "string", "title": "Rewrite" }, "case_sensitive": { "type": "boolean", "title": "Case Sensitive", "default": false } }, "type": "object", "required": [ "original", "rewrite" ], "title": "PronunciationRuleCreate" }, "PronunciationRuleResponse": { "properties": { "original": { "type": "string", "title": "Original" }, "rewrite": { "type": "string", "title": "Rewrite" }, "case_sensitive": { "type": "boolean", "title": "Case Sensitive", "default": false }, "id": { "type": "integer", "title": "Id" } }, "type": "object", "required": [ "original", "rewrite", "id" ], "title": "PronunciationRuleResponse", "description": "Pronunciation rule response schema." }, "ValidationError": { "properties": { "loc": { "items": { "anyOf": [ { "type": "string" }, { "type": "integer" } ] }, "type": "array", "title": "Location" }, "msg": { "type": "string", "title": "Message" }, "type": { "type": "string", "title": "Error Type" } }, "type": "object", "required": [ "loc", "msg", "type" ], "title": "ValidationError" }, "VoiceCreateResponse": { "properties": { "uid": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Uid" }, "error": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Error" }, "was_updated": { "type": "boolean", "title": "Was Updated", "default": false } }, "type": "object", "title": "VoiceCreateResponse" }, "VoiceResponse": { "properties": { "uid": { "type": "string", "title": "Uid" }, "name": { "type": "string", "title": "Name" }, "description": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Description" }, "language": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Language" }, "start_s": { "type": "number", "title": "Start S" }, "stop_s": { "anyOf": [ { "type": "number" }, { "type": "null" } ], "title": "Stop S" }, "filename": { "type": "string", "title": "Filename" }, "org_uid": { "anyOf": [ { "type": "string", "format": "uuid" }, { "type": "null" } ], "title": "Org Uid" }, "is_pending": { "type": "boolean", "title": "Is Pending", "default": false }, "has_audio": { "type": "boolean", "title": "Has Audio", "default": true }, "is_pro_clone": { "type": "boolean", "title": "Is Pro Clone", "default": false } }, "type": "object", "required": [ "uid", "name", "start_s", "filename" ], "title": "VoiceResponse", "description": "Schema for voice response data." }, "VoiceUpdate": { "properties": { "name": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Name" }, "description": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Description" }, "language": { "anyOf": [ { "type": "string" }, { "type": "null" } ], "title": "Language" }, "start_s": { "anyOf": [ { "type": "number" }, { "type": "null" } ], "title": "Start S" }, "tags": { "anyOf": [ { "items": { "additionalProperties": { "anyOf": [ { "type": "string" }, { "type": "boolean" }, { "type": "null" } ] }, "type": "object" }, "type": "array" }, { "type": "null" } ], "title": "Tags" }, "rank": { "anyOf": [ { "type": "number" }, { "type": "null" } ], "title": "Rank" } }, "type": "object", "title": "VoiceUpdate", "description": "Schema for updating voice data." } } }, "x-tagGroups": [ { "name": "Documentation", "tags": [ "Documentation", "FAQ", "Release notes" ] }, { "name": "API Reference", "tags": [ "TTS", "STT", "Voices", "Pronunciations", "Credits" ] } ], "tags": [ { "name": "Documentation", "description": "# Features\n\n- **Multilingual**: We currently support five languages: English (en), French (fr), German (de), Spanish (es) and Portuguese (pt) for our Text-To-Speech and Speech-To-Text with more languages to come. \n- **Low-latency**: Our servers are based in Europe and in the US, with our expected time-to-first-token is below 300ms when streaming.\n- **Voice selection**: We provide a voice library, with multiple voices to choose from in different languages. You can also clone voices instantaneously using a 10'' voice sample. \n\n# Installation\n\n```bash\npip install gradium\n```\n\n# Quick Start\n\n```python\nimport asyncio\nimport gradium\n\nasync def main():\n client = gradium.client.GradiumClient(api_key=\"your-api-key\")\n\n result = await client.tts(\n setup={\"voice_id\": \"YTpq7expH9539ERJ\", \"output_format\": \"wav\"},\n text=\"Welcome to Gradium! Transform your text into natural-sounding speech in seconds.\"\n )\n\n with open(\"welcome.wav\", \"wb\") as f:\n f.write(result.raw_data)\n\nif __name__ == \"__main__\":\n asyncio.run(main())\n```\n\n# Creating a Client\n\n## Using API Key Directly\n\n```python\nimport gradium\n\nclient = gradium.client.GradiumClient(api_key=\"gd_your_api_key_here\")\n```\n\n## Using Environment Variable\n\nSet the `GRADIUM_API_KEY` environment variable:\n\n```bash\nexport GRADIUM_API_KEY=gd_your_api_key_here\n```\n\nThen create the client without passing the API key:\n\n```python\nclient = gradium.client.GradiumClient()\n```\n\n# Text-to-Speech (TTS)\n\n## Basic Usage\n\n```python\nimport gradium\n\nclient = gradium.client.GradiumClient()\nresult = await client.tts(\n setup={\n \"model_name\": \"default\", \n \"voice_id\": \"YTpq7expH9539ERJ\",\n \"output_format\": \"wav\"\n },\n text=\"Hello, world!\"\n)\n\nwith open(\"output.wav\", \"wb\") as f:\n f.write(result.raw_data)\n\nprint(f\"Sample rate: {result.sample_rate}\")\nprint(f\"Request ID: {result.request_id}\")\n```\n\n## Setup Parameters\n\n- **`model_name`**: The TTS model to use (default: `\"default\"`)\n- **`voice_id`**: The voice id of the voice to be used. The voice id can be found in the voice library section of this documentation or in the studio.\n- **`output_format`**: Audio format of the input data (supported: `\"pcm\"`,\n `\"wav\"`, `\"opus\"`, ...)\n\nWhen using `\"pcm\"` output format, the audio will adhere to the following\nspecifications:\n- **Sample Rate**: 48000 Hz (48kHz)\n- **Format**: PCM (Pulse Code Modulation)\n- **Bit Depth**: 16-bit signed integer\n- **Channels**: Single channel (mono)\n- **Chunk Size**: 3840 samples per chunk (80ms at 48kHz)\n\nWhen using the `\"wav\"` output format, the audio chunks are in WAV format,\nat 48kHz, 16-bit signed integer mono.\n\nWhen using the `\"opus\"` output format, the audio chunks use the Opus codec\nwrapped in an Ogg container.\n\nAlternative output formats include `\"ulaw_8000\"`, `\"alaw_8000\"`, `\"pcm_8000\"`,\n`\"pcm_16000\"`, and `\"pcm_24000\"`.\n\n\n## Streaming TTS\n\nThe TTS can be used in a streaming fashion. The first chunks of audio will be\navailable as soon as they are generated.\n\n```python\nstream = await client.tts_stream(\n setup={\n \"model_name\": \"default\",\n \"voice_id\": \"LFZvm12tW_z0xfGo\",\n \"output_format\": \"pcm\"\n },\n text=\"This is a longer text that will be streamed.\"\n)\n\nasync for audio_chunk in stream.iter_bytes():\n print(f\"Received {len(audio_chunk)} bytes\")\n```\n\n## Using Custom Voices\n\n```python\nresult = await client.tts(\n setup={\n \"model_name\": \"default\",\n \"voice_id\": \"YTpq7expH9539ERJ\",\n \"output_format\": \"wav\"\n },\n text=\"Hello with my custom voice!\"\n)\n```\n\n## Output Formats\n\n```python\n# WAV format\nresult = await client.tts(setup={\"voice_id\": \"YTpq7expH9539ERJ\", \"output_format\": \"wav\"}, text=\"Hello\")\n\n# PCM format: the data is sampled at 48kHz, 16-bit signed integer, mono\nresult = await client.tts(setup={\"voice_id\": \"YTpq7expH9539ERJ\", \"output_format\": \"pcm\"}, text=\"Hello\")\n\n# Get numpy array from PCM\npcm_array = result.pcm()\npcm16_array = result.pcm16()\n```\n\n## Flushing and Pauses\n\nThe model only generates audio when it has enough context to do so, so generally\nthe audio lags a few words behind the text input. The `` tag can be\nused to force the model to output the audio for all the text that has been\ninput so far.\n\n```python\nsample_text = \"Hello, this is a test from the Gradium Text to Speech system. We are testing the flush.\"\ntest_audio = await client.tts(\n setup={'voice_id': 'YTpq7expH9539ERJ', 'output_format': 'wav'},\n text=sample_text,\n)\n```\n\nPauses can be generated by inserting a \"break time\" tag as show below.\nThe brak time is specified in seconds and should be between 0.1 and 2.0s.\nThe tag must be preceeded and followed by a space.\n\n```python\n\nsample_text = 'Hello, this is a test from the Gradium Text to Speech system. We are testing the pause.'\n\ntest_audio = await client.tts(\n setup={'voice_id': 'YTpq7expH9539ERJ', 'output_format': 'wav'},\n text=sample_text,\n)\n```\n\n## Text with Timestamps\n\nThe model also returns word-level timestamps for the generated audio.\n\n```python\nresult = await client.tts(\n setup={\"voice_id\": \"YTpq7expH9539ERJ\", \"output_format\": \"wav\"},\n text=\"Hello, world!\"\n)\n\nfor item in result.text_with_timestamps:\n print(f\"{item.text}: {item.start_s:.2f}s - {item.stop_s:.2f}s\")\n```\n\n\n## Async Generator Input\n\n```python\nasync def text_generator():\n yield \"Hello, \"\n yield \"this is \"\n yield \"a streaming \"\n yield \"example.\"\n\nstream = await client.tts_stream(\n setup={\"voice_id\": \"YTpq7expH9539ERJ\", \"output_format\": \"pcm\"},\n text=text_generator()\n)\n\nasync for chunk in stream.iter_bytes():\n pass\n```\n\n## Pronunciation Dictionaries\n\nPronunciation dictionaries allow you to customize how specific words or phrases are pronounced in your TTS output. This is particularly useful for:\n- Brand names, technical terms, or proper nouns\n- Acronyms that should be pronounced in a specific way\n- Words with non-standard pronunciations in your use case\n\n\nThe easiest way to create and manage pronunciation dictionaries is through the Gradium Studio, on the pronunciation page.\nOnce you have created a dictionary and obtained its ID, you can use it in your TTS requests by passing the `pronunciation_id` parameter in the setup message, similar to the way we pass the voice_id:\n\n```python\nimport gradium\n\nclient = gradium.client.GradiumClient()\n\nresult = await client.tts(\n setup={\n \"voice_id\": \"YTpq7expH9539ERJ\",\n \"output_format\": \"wav\",\n \"pronunciation_id\": \"bb1ckYhNHCcIJjdK\", # Whatever your ID is.\n },\n text=\"The text you want to generate.\"\n)\n\nwith open(\"output.wav\", \"wb\") as f:\n f.write(result.raw_data)\n```\n\n# Multiplexing\n\nMultiplexing allows you to send multiple independent TTS requests over a single WebSocket connection. Each request is tracked independently using a unique identifier, allowing concurrent processing of multiple text inputs without opening multiple connections.\n\nWhen multiplexing is enabled, each message you send to the server must include a `client_req_id` field. The server will stamp all response messages (audio chunks, metadata, etc.) with the same `client_req_id`, allowing you to match responses to their corresponding requests.\n\nTo enable multiplexing, include `close_ws_on_eos: False` in your setup message. This tells the server to keep the WebSocket connection open after completing individual requests.\n\n```python\nsetup = {\n \"voice_id\": \"RI2y7oBdsQJmkgFF\",\n \"output_format\": \"wav\",\n \"close_ws_on_eos\": False # Enable multiplexing\n}\ntexts = [\n \"First request. Second part, last one.\",\n \"Second request. Second part, last one again.\",\n]\n\nclient = gradium.client.GradiumClient(base_url=url, api_key=api_key)\nasync with client.tts_realtime(send_setup_on_start=False) as stream:\n \n async def send_loop():\n for idx, text in enumerate(texts):\n stamp = {'client_req_id': f'req-{idx:02d}'}\n await stream.send_setup(setup | stamp)\n await stream.send_text(text, **stamp)\n await stream.send_eos(**stamp)\n \n async def recv_loop():\n audio = collections.defaultdict(list)\n num_eos = 0\n async for msg in stream:\n if msg[\"type\"] == \"audio\":\n audio[msg.get('client_req_id')].append(msg[\"audio\"])\n elif msg['type'] == 'end_of_stream':\n num_eos += 1\n if num_eos == len(texts):\n break\n return audio\n\n _, audio = await asyncio.gather(send_loop(), recv_loop())\n audio = {k: b\"\".join(v) for k, v in audio.items()}\n```\n\n\n# Advanced Options\n\nSome models support advanced options that can be passed using the `json_config`\nparameter. In the Python api, this parameter is passed as a dictionary mapping\nstring to values (either float or string).\n\nThis parameter can be used to control:\n- Speed of the generated speech via the `padding_bonus` parameter.\n- Stability of the generated speech via the `temp` temperature parameter.\n- Voice similarity using the `cfg_coef` parameter.\n- Rewrite rules using `rewrite_rules`.\n\n**Speed Control** You can guide the speed of the model using the padding bonus\nparameter. Default value is 0.0. Negative values mean that the speaker will speak\nfaster (values between -4.0 and -0.1) Positive values mean that the speaker will\nspeak slower (values between 0.1 and 4.0)\n\n```python\nsample_text = \"Hello, this is a test from the Gradium Text to Speech system. We are testing the speed.\"\n\nslower_audio = await client.tts(\n setup={'voice_id': 'YTpq7expH9539ERJ', 'output_format': 'wav', 'json_config':{'padding_bonus':2.0}},\n text=sample_text,\n)\n\nfaster_audio = await client.tts(\n setup={'voice_id': 'YTpq7expH9539ERJ', 'output_format': 'wav', 'json_config':{'padding_bonus':-2.0}},\n text=sample_text,\n)\n```\n\n**Temperature Control** The temperature for the generation can be set with\nvalues ranging from 0 to 1.4. A value of 0 corresponds to a deterministic\ngeneration, while higher values lead to more diverse outputs. Default value is\n0.7.\n\n```python\nsetup = {'voice_id': 'YTpq7expH9539ERJ', 'output_format': 'wav', 'json_config':{'temp':0.3}}\n\naudio = await client.tts(text=sample_text, setup=setup)\n```\n\n**Voice Similarity Control** The `cfg_coef` parameter can be used to control the\nsimilarity of the generated speech to the target voice. Values range from 1.0 to\n4.0. The default value is 2.0. The higher the value, the more the model\nreplicates the cloned voice but larger values can lead to audio artifacts.\n\n**Rewrite Rules** The `rewrite_rules` parameter can be used to pass text\nrewriting rules that are applied before the text is synthesized. The rules\nshould be passed as a string. More details on the rules themselves can be found\nbelow in this document. Values such as `\"en\"`, `\"fr\"`, `\"de\"`, `\"es\"`, `\"pt\"`\nenable all the rewriting rules for a given language.\n\n# Voices Library\n\nGradium provides a selection of high-quality voices across multiple languages. Here are our voices.\n\n## Flagship Voices\n\n\n| Name | Voice ID            | Language | Country | Age Group | Gender | Description |\n| :--- | :--- | :---: | :---: | :---: | :---: | :--- |\n| **Emma** | `YTpq7expH9539ERJ` | `en` | `us` 🇺🇸 | Adult | Feminine | A pleasant and smooth female voice ready to assist your customers and also eager to have nice conversations. |\n| **Kent** | `LFZvm12tW_z0xfGo` | `en` | `us` 🇺🇸 | Adult | Masculine | A relaxed and authentic American adult voice that connects like a genuine friend. |\n| **Sydney** | `jtEKaLYNn6iif5PR` | `en` | `us` 🇺🇸 | Adult | Feminine | A joyful and airy American adult voice that makes corporate training feel helpful and light.|\n| **John** | `KWJiFWu2O9nMPYcR` | `en` | `us` 🇺🇸 | Adult | Masculine | A warm low-pitched American adult voice with the resonant quality of a classic radio broadcaster. |\n| **Eva** | `ubuXFxVQwVYnZQhy` | `en` | `gb` 🇬🇧 | Adult | Feminine | A joyful and dynamic British adult voice ideal for lively conversations. |\n| **Jack** | `m86j6D7UZpGzHsNu` | `en` | `gb` 🇬🇧 | Adult | Masculine | A pleasant British voice suited for helpful service, casual conversations, or intense narrations. |\n| **Elise** | `b35yykvVppLXyw_l` | `fr` | `fr` 🇫🇷 | Adult | Feminine | A warm and smooth French adult voice ideal for friendly conversation and welcoming support. |\n| **Leo** | `axlOaUiFyOZhy4nv` | `fr` | `fr` 🇫🇷 | Adult | Masculine | A warm and smooth French adult voice ideal for friendly conversation and welcoming support. |\n| **Mia** | `-uP9MuGtBqAvEyxI` | `de` | `de` 🇩🇪 | Adult | Feminine | A joyful and energetic German voice perfect for professional context as well as enthusiastic discussions. |\n| **Maximilian** | `0y1VZjPabOBU3rWy` | `de` | `de` 🇩🇪 | Adult | Masculine | A warm and smooth German adult voice ideal for friendly conversation and professional narration. |\n| **Valentina** | `B36pbz5_UoWn4BDl` | `es` | `mx` 🇲🇽 | Adult | Feminine | A warm and engaging Mexican female voice perfect for natural storytelling and connecting like a genuine friend. |\n| **Sergio** | `xu7iJ_fn2ElcWp2s` | `es` | `es` 🇪🇸 | Adult | Masculine | A warm and smooth Spanish adult voice ideal for friendly conversation and professional narration. |\n| **Alice** | `pYcGZz9VOo4n2ynh` | `pt` | `br` 🇧🇷 | Adult | Feminine | A warm and smooth Brazilian female voice ideal for professional service and pleasant narration or even an enthusiastic conversation! |\n| **Davi** | `M-FvVo9c-jGR4PgP` | `pt` | `br` 🇧🇷 | Adult | Masculine | An engaging and smooth Brazilian adult voice ideal for helpful service and relaxing conversations. |\n\n\n## All Voices \n\n
\n View all voices\n\nName | voice_id | Language | Country | Perceived age | Perceived gender | Description\n| :--- | :--- | :--- | :--- | :--- | :--- | :---\nEva | ubuXFxVQwVYnZQhy | en | gb | Adult | Feminine | A joyful and dynamic British adult voice ideal for lively conversations\nJack | m86j6D7UZpGzHsNu | en | gb | Adult | Masculine | A pleasant British voice suited for helpful service casual conversations or intense narrations\nEmma | YTpq7expH9539ERJ | en | us | Adult | Feminine | A pleasant and smooth female voice ready to assist your customers and also eager to have nice converstations\nKent | LFZvm12tW_z0xfGo | en | us | Adult | Masculine | A relaxed and authentic American adult voice that connects like a genuine friend.\nMia | -uP9MuGtBqAvEyxI | de | de | Adult | Feminine | A joyful and energetic German voice perfect for professional context as well as enthusiastic discussions.\nMaximilian | 0y1VZjPabOBU3rWy | de | de | Adult | Masculine | A warm and smooth German adult voice ideal for friendly conversation and professional narration.\nValentina | B36pbz5_UoWn4BDl | es | mx | Adult | Feminine | A warm and engaging Mexican female voice perfect for natural storytelling and connecting like a genuine friend.\nSergio | xu7iJ_fn2ElcWp2s | es | es | Adult | Masculine | A warm and smooth Spanish adult voice ideal for friendly conversation and professional narration.\nElise | b35yykvVppLXyw_l | fr | fr | Adult | Feminine | A warm and smooth French adult voice ideal for friendly conversation and welcoming support.\nLeo | axlOaUiFyOZhy4nv | fr | fr | Adult | Masculine | A warm and smooth French adult voice ideal for friendly conversation and welcoming support.\nAlice | pYcGZz9VOo4n2ynh | pt | br | Adult | Feminine | A warm and smooth Brazilian female voice ideal for professional service and pleasant narration or even an enthusiastic conversation!\nDavi | M-FvVo9c-jGR4PgP | pt | br | Adult | Masculine | An engaging and smooth Brazilian adult voice ideal for helpful service and relaxing conversations.\nMax | NoJdNY6JTz-VJLwz | en | ca | Young Adult | Masculine | A clear calm and measured male voice.\nKelly | Lxc7YlPC8ckLJA8H | en | gb | Adult | Feminine | Clear soft and measured female narration.\nArjun | -_aUUFZaJ0CT1gks | en | in | Adult | Masculine | A warm voice with a clear low-pitch and a smooth texture.\nHunter | W5htOuyiFI4Fwhxs | en | au | Adult | Masculine | A joyful and smooth Australian adult voice that keeps listeners tuned in with radio charm.\nTiffany | Eu9iL_CYe8N-Gkx_ | en | us | Young Adult | Feminine | A warm and smooth American young adult voice that greets customers with a smile you can hear.\nChristina | 2H4HY2CBNyJHBCrP | en | us | Adult | Feminine | A joyful low-pitched American adult voice that handles business and service with efficiency.\nMaria | KNYHZTB8ZqdAZv5Q | en | us | Adult | Feminine | A joyful high-pitched American adult voice that teaches and tutors with genuine energy.\nMark | dh0EzP6jCroK6prq | en | us | Adult | Masculine | A warm low-pitched American adult voice that resonates with professional radio quality.\nLogan | XJc-Y9tkSd1UA7s4 | en | us | Young Adult | Masculine | A joyful and smooth American young adult voice that fits the energetic vibe of a gym coach.\nJuan | 78zAgQK6xmExb8wS | en | us | Adult | Masculine | A joyful and smooth American adult voice that welcomes and hosts with vibrant energy.\nKaitlyn | 56DcpvEI0Gawpidh | en | us | Adult | Feminine | A warm and smooth American adult voice that offers the kindness of a helpful neighbor.\nMichelle | lt88kyLfD8Mqemla | en | in | Young Adult | Feminine | A warm and smooth Indian English young adult voice for clear and friendly service.\nMary | wPx6HPbUQkaUHGhq | en | us | Adult | Feminine | A joyful high-pitched American adult voice that connects perfectly with younger audiences.\nCameron | c8BzreHTk1GG2R4z | en | us | Adult | Masculine | A steady low-pitched American adult voice ideal for tech reviews and casual explanations.\nJeremy | 9QHzSiOYUD-RzEzM | en | us | Adult | Masculine | A composed American adult voice that sounds intelligent and tech-savvy.\nJesse | hOhCtzjR-cRG4T5T | en | us | Young Adult | Masculine | A joyful high-pitched American young adult voice with a unique airy texture for character roles.\nSean | cu0XE3Cxmg_GmSJ3 | en | us | Adult | Masculine | A joyful American adult voice that brings the spirited energy of a rodeo announcer.\nCharles | P0GYBrxlhTy5CC87 | en | gb | Adult | Masculine | A warm and smooth British adult voice that hosts with a classic reliable radio presence.\nOlivia | kr-Om35JRqmA3Hzq | en | us | Young Adult | Feminine | A warm low-pitched American young adult voice that guides meditation with soothing calm.\nShelby | O0uTTRx5zcetDFX4 | en | us | Young Adult | Feminine | A joyful high-pitched American young adult voice that brings enthusiasm to web content.\nPatrick | Z5GIOZR45ieZ8M-W | en | us | Adult | Masculine | A joyful and smooth American adult voice perfect for clear and engaging public service announcements.\nRichard | HndphaVV7KTCfKQT | en | us | Adult | Masculine | A joyful high-pitched American adult voice that captures the excitement of sports commentary.\nJason | FOFDH8py3aghc5kb | en | us | Adult | Masculine | A joyful American adult voice that delivers radio content with a distinct engaging tone.\nKimberly | Abqwk2RWxlBEyv0j | en | gb | Adult | Feminine | A joyful high-pitched British adult voice that welcomes listeners with cheerful efficiency.\nTimothy | v5lib8tjaosy5sxQ | en | us | Adult | Masculine | A warm low-pitched American adult voice with a nostalgic friendly resonance.\nNathan | 4NU5PqxX2BdMEtWe | en | us | Adult | Masculine | A warm and smooth American adult voice that sounds just like your friendly neighbor.\nAdam | EbIA5CIcQoa6NNd2 | en | us | Adult | Masculine | A joyful and smooth American adult voice that greets the morning with radio-ready energy.\nAbigail | KRo-uwfno-KcEgBM | en | us | Adult | Feminine | A warm and airy American adult voice that adds a touch of magic and empathy to any story.\nMelissa | 8Tm8RKFEbnkRtkdA | en | us | Adult | Feminine | A joyful and smooth American adult voice that facilitates with upbeat enthusiasm.\nAllison | yU6yxQ3e8LKRwU84 | en | us | Adult | Feminine | A joyful high-pitched American adult voice that brings high energy to training and teaching.\nKelsey | MQC0U1yWvZXrppaF | en | us | Adult | Feminine | A balanced American adult voice that fits realistic everyday service interactions.\nHaley | aq7ltaIQ6ZJUY0jR | en | gb | Adult | Feminine | A confident and warm British adult voice versatile enough for e-learning support and storytelling.\nAnna | PS7enm5lVZiIvEKV | en | us | Adult | Feminine | A warm and smooth American adult voice that provides comfort and supportive guidance.\nKatherine | bvNlBZ3DWDoVy_Yc | en | us | Young Adult | Feminine | A warm and smooth American young adult voice that balances business professionalism with kindness.\nSteven | zyLIanWKViHkc6Wp | en | gb | Adult | Masculine | A steady and smooth British adult voice that offers helpful and consistent management advice.\nBrian | ptMwY_gvmFxXMmDf | en | us | Adult | Masculine | A steady American adult voice with a low-pitched tone suitable for distinct character roles.\nJose | LqFNS0u6EII7VHBx | en | us | Adult | Masculine | A warm low-pitched American adult voice that offers the reassuring guidance of a mentor.\nMadison | cuXxqSrGVntdhFpZ | en | gb | Young Adult | Feminine | A warm low-pitched British young adult voice that feels like a friendly neighbor.\nDylan | d9Fl9x8luXXX7u6E | en | us | Adult | Masculine | A warm and smooth American adult voice that keeps the flow going as a DJ or host.\nRebecca | GJSxJhSTPAGIPDwy | en | us | Adult | Feminine | A warm and airy American adult voice that manages and assists with a gentle touch.\nSamuel | pxKsJ_4kEMid5XpZ | en | au | Young Adult | Masculine | A warm and smooth Australian young adult voice that sounds like a friendly bartender or colleague.\nEric | knw-ddWDPNORRA4Z | en | us | Adult | Masculine | A joyful and smooth American adult voice that makes sales and service feel cheerful and easy.\nAlyssa | 22YWyuFACaMHsPh5 | en | us | Young Adult | Feminine | A warm and smooth American young adult voice that adds a relatable human touch to readings.\nAlexandra | 4nAcNUlNhEA_Kyjo | en | us | Adult | Feminine | A joyful and smooth American adult voice ideal for reading and hosting duties.\nJasmine | QPHuXnvRPQ57oXYy | en | us | Adult | Feminine | A joyful high-pitched American adult voice that commands the room with managerial confidence.\nBenjamin | IBVzgY91NZ1IJ0oP | en | gb | Adult | Masculine | A joyful and smooth British adult voice that leads events and shows with master-of-ceremony flair.\nAaron | Ve1zknlflaRwcAQw | en | gb | Adult | Masculine | A composed and smooth British adult voice perfect for technical and IT-related explanations.\nJordan | ws0Wb0PZXl21_Bbz | en | us | Young Adult | Masculine | A joyful American young adult voice that motivates with the energy of a fitness instructor.\nChristian | x69x43aS-5mVLCX2 | en | gb | Adult | Masculine | A warm and smooth British adult voice that sounds like a kind and knowledgeable scholar.\nThomas | m7fJRmVaJjG2TL1c | en | gb | Adult | Masculine | A warm and smooth British adult voice that brings an actor's versatility to conversation.\nMorgan | MGiwMOFxVe4a2aSU | en | gb | Adult | Feminine | A warm and airy British adult voice that guides listeners into a state of meditation.\nCody | SqHUVuEiTPSlIB5r | en | us | Adult | Masculine | A warm and resonant American adult voice that delivers radio quality with a professional touch.\nAlex | 91EdXxJDbWICDBgz | en | us | Adult | Neutral | A joyful high-pitched American adult voice that grabs attention in advertisements.\nBrianna | fggSYM_FGJ30QTTl | en | us | Young Adult | Feminine | A warm and smooth American young adult voice ideal for music radio and educational content.\nKevin | J2qsArcdozbto5Hn | en | au | Adult | Masculine | A joyful Australian adult voice that engages audiences as a TV host or tutor.\nVictoria | 8dBmiTurwb7KcxLY | en | us | Adult | Feminine | A warm and smooth American adult voice that conveys the reliability of a helpful colleague.\nNicole | T7UL6gmeDqqYiVe1 | en | us | Adult | Feminine | A joyful American adult voice with a sarcastic edge perfect for entertaining podcasts.\nJennifer | auZu0iT-fniQ4cJd | en | us | Adult | Feminine | A warm and smooth American adult voice that is always ready to help like a good friend.\nCourtney | UX3Hi2ZmK7tT0c3G | en | gb | Adult | Feminine | A joyful high-pitched British adult voice perfect for sales and professional announcements.\nStephanie | ikbJkd83GvuyoSLb | en | us | Adult | Feminine | A joyful and smooth American adult voice that sounds like a modern relatable mom.\nKyle | CjQcj4yeIs6h0uAb | en | us | Adult | Masculine | A joyful American adult voice that wakes up the audience with morning radio energy.\nLauren | SG3KnxbSOkkrY097 | en | us | Adult | Feminine | An assertive and smooth American adult voice that fits the modern urban businesswoman persona.\nAlexis | 74asmf7CXzjfopIX | en | us | Adult | Feminine | A joyful American adult voice that delivers customer service scripts with a bright distinct tone.\nMegan | exG4bLr-lZ_bI0jF | en | us | Adult | Feminine | A joyful high-pitched American adult voice that mixes customer service clarity with influencer energy.\nJonathan | 4u2uvwrHdTA2gRnZ | en | us | Adult | Masculine | A joyful high-pitched American adult voice with the charm of an old-timey character actor.\nRobert | gTAO-3xLZ8_WSfbm | en | us | Adult | Masculine | A warm and resonant American adult voice that brings a professional acting polish to any script.\nAlexander | 8sWSyTC7byLsbHkr | en | us | Adult | Masculine | A warm low-pitched American adult voice that motivates with the resonance of a fitness coach.\nRachel | dEcrv3B8XGHoox2_ | en | gb | Adult | Feminine | A warm low-pitched British adult voice that balances professional business tones with a calming presence.\nKayla | 9VXl5t2IMagUQAzg | en | gb | Adult | Feminine | A joyful British adult voice with a precise tone ideal for automated yet friendly service.\nElizabeth | u8rA2xOF_0LRnNSb | en | us | Adult | Feminine | A consistent and smooth American adult voice that provides clear and reliable customer service.\nAmanda | ZZb4X9ueHSdRlv9q | en | gb | Young Adult | Feminine | A joyful and hip British young adult voice that brings energy to podcasts and modern content.\nBrittany | 3bIdO9CHnAh_pRAf | en | in | Adult | Feminine | A joyful and smooth Indian English adult voice that is perfect for friendly HR and service roles.\nWilliam | VeVmpxxbyJiWrGNG | en | au | Adult | Masculine | A joyful high-pitched Australian adult voice that sounds like an energetic high school coach.\nHannah | lP7D1y02OQFtffU3 | en | us | Young Adult | Feminine | A warm and airy American young adult voice that creates a calm atmosphere for yoga and meditation.\nAnthony | 2V3TjbyQGPlkY6ON | en | au | Adult | Masculine | A joyful and smooth Australian adult voice that brings a cartoonish MC-style energy.\nJustin | 6Mp6PGnaCdb-US21 | en | us | Adult | Masculine | A distinct American adult voice with a characterful tone ideal for niche roles.\nJames | MZWrEHL2Fe_uc2Rv | en | us | Adult | Masculine | A warm and resonant American adult voice that excels at storytelling and persuasive advertising.\nDavid | OceLYI_PPbqsdgdV | en | gb | Young Adult | Masculine | A warm and smooth British young adult voice that captures the relaxed tone of a college student.\nRyan | AqRuVz8-e8u3BR00 | en | us | Adult | Masculine | A warm low-pitched American adult voice with a resonant rural charm for sales and storytelling.\nTaylor | EfuzJVuTmw_mA7PC | en | us | Adult | Feminine | A warm and efficient American adult voice that fits perfectly for automated customer support.\nSarah | aW5dxfdkzIFCIdXc | en | us | Young Adult | Feminine | A clear American young adult voice that is precise and perfect for student-focused reading.\nJoseph | MhsYZQ4bIfcDpokF | en | gb | Adult | Masculine | A warm and relatable British adult voice with a genuine blue-collar friendliness.\nSamantha | mn5sS7D8kYKETZXA | en | us | Adult | Feminine | A warm and professional American adult voice that is both helpful and authoritatively managerial.\nAustin | -0MuXG9RcCsuSVtb | en | us | Mature | Masculine | A warm rough-textured American mature voice that embodies the kindness of a gentle grandfather.\nDaniel | apU2CMobTyu92tZj | en | au | Adult | Masculine | A joyful and smooth Australian adult voice that brings a cheerful down-to-earth vibe to any chat.\nEmily | i1kmq28cO60ia35K | en | us | Young Adult | Feminine | A warm and smooth American young adult voice perfect for modern podcasting and influencing.\nBrandon | 2j8TWGsIiUl4G3kj | en | us | Young Adult | Masculine | A high-pitched joyful American young adult voice that sounds like your friendliest colleague.\nTyler | Ow5IKhni2ED3Xxhl | en | gb | Adult | Masculine | A warm and smooth British adult voice that blends tech-savviness with a friendly radio persona.\nNicholas | n2Gv34jje2ZiiNzK | en | us | Adult | Masculine | A joyful American adult voice with a relatable slightly clumsy charm perfect for sitcom-style scripts.\nAshley | QZMzHBlnJRjll_71 | en | us | Adult | Feminine | A warm low-pitched American adult voice that feels like a cool supportive friend or aunt.\nJoshua | bDlMqRew31ZJwrD- | en | us | Adult | Masculine | A joyful and resonant American adult voice that brings the classic energy of a radio host.\nJessica | wYY8mXKrKtwKsaXZ | en | us | Adult | Feminine | A consistent and smooth American adult voice that handles customer service with patience and clarity.\nJacob | ixaCTlZ5Xqf2XzQH | en | us | Mature | Masculine | A steady American mature voice with a unique old-timey texture for distinct conversational roles.\nChristopher | fs2Qj_X2Z2WvWJSU | en | gb | Adult | Masculine | A smooth British adult voice that conveys the trustworthy tone of a reliable expert.\nMatthew | X-wgJsZwQKhfebgK | en | us | Adult | Masculine | A high-pitched joyful American adult voice that pops with energy perfect for reading ads.\nMichael | Mj0Pzs94jCw8oVOC | en | us | Adult | Masculine | A low-pitched casual American adult voice with a sporty vibe for conversational content.\nOlivier | vMYQUSzm6GRkJX6d | fr | fr | Adult | Masculine | Friendly male voice tone is warm and welcoming.\nManon | p1fSBpcmVWngBqVd | fr | fr | Young Adult | Feminine | A gentle and warm voice with a calm and measured pace.\nJade | 3mM3xaoFjNMQa22C | fr | fr | Young Adult | Feminine | A young female speaker with a clear high-pitched and smooth voice.\nAmélie | J4XbCGPYNMigXcfZ | fr | fr | Young Adult | Feminine | A friendly voice with a clear tone and pleasant pitch.\nAdrien | 0LMAi0x_YVG_GLeM | fr | fr | Young Adult | Masculine | Clear smooth and moderately paced voice with a warm tone.\nSarah | -dOnYAX4N4GqSOee | fr | fr | Young Adult | Feminine | A warm and smooth French young adult voice perfect for friendly interactions and welcoming service.\nJennifer | N8xxxD_d-ZinGVI4 | fr | fr | Young Adult | Feminine | A warm and smooth French young adult voice ideal for friendly support and welcoming conversation.\nÉlodie | zba0owtqy4Gnewn9 | fr | fr | Adult | Feminine | A confident French adult voice that excels in corporate training compliance and narration.\nJustine | TJv-kucMsUo24VQe | fr | fr | Young Adult | Feminine | A confident and upbeat French young adult voice perfect for youth brands and energetic explanations.\nOcéane | YE0-JPiElafJrZaC | fr | fr | Young Adult | Feminine | A polished French young adult voice designed for professional broadcasting and reporting.\nLéa | QY_BJKHMElKDO12- | fr | fr | Adult | Feminine | A formal French adult voice that delivers financial reports and news with absolute precision.\nSarah | QkmUhBH4hIV2_BkY | fr | fr | Adult | Feminine | A confident and compassionate French adult voice ideal for biographies support and non-fiction.\nMathieu | D-IpHY1UI0iX9xQD | fr | fr | Adult | Masculine | An assertive and energetic French adult voice perfect for high-stakes promos and executive presentations.\nClément | twLGV8mrH_ycNpUn | fr | fr | Adult | Masculine | A confident and sincere French adult voice that lends credibility to expert topics and emotional appeals.\nJulie | k1wgs3k8-wRxTJO6 | fr | fr | Adult | Feminine | A joyful and enthusiastic French adult voice that makes news and education feel fresh and engaging.\nDylan | Hdf5cdfaGrLDTD63 | fr | fr | Adult | Masculine | A sincere and emotional French adult voice that offers genuine support and relatable warmth.\nMarion | 1VAVLmmbQFDw7TMn | fr | fr | Adult | Feminine | A warm and trustworthy French adult voice that shines in storytelling education and fantasy roles.\nPauline | 2AtP1urAQkZaeI2U | fr | fr | Adult | Feminine | A professional and articulate French adult voice suited for serious journalism and formal announcements.\nVincent | B09t5S64xLaKwXeW | fr | fr | Adult | Masculine | A warm and wise French adult voice perfect for historical narration and supportive guidance.\nPierre | AroCL6f1qizjiZ_a | fr | fr | Young Adult | Masculine | An energetic French young adult voice that brings a lively journalistic flair to news and updates.\nGuillaume | qTA0lxFpynJdoxx7 | fr | fr | Young Adult | Masculine | A joyful and adventurous French young adult voice ideal for dynamic storytelling and sports reporting.\nRomain | zpmn3GOfiU_i5QGo | fr | fr | Adult | Masculine | A warm and steady French adult voice that delivers quick instructions and interviews with clarity.\nKévin | IB53xJtufx1sbfbt | fr | fr | Adult | Masculine | A sincere and emotional French adult voice that brings depth and wisdom to narratives and heartfelt ads.\nFlorian | kw_VWSocR7vyA9Ty | fr | fr | Adult | Masculine | A joyful and relatable French adult voice that sounds like a friendly journalist or the guy next door.\nAntoine | hx1RAC4Lqd9xyTAr | fr | fr | Adult | Masculine | A gritty and confident French adult voice perfect for intense narration and expert instruction.\nQuentin | pdcyd1mLmo0fcg3O | fr | fr | Adult | Masculine | A confident and sincere French adult voice that connects effortlessly in tech explainers and documentaries.\nMélanie | xynYWquoAsrvM7UY | fr | ca | Adult | Feminine | A warm and clear Canadian French adult voice designed for friendly assistance and educational guidance.\nAdam | aNiSRZ0BhQxO1FPx | fr | fr | Adult | Masculine | A warm and formal French adult voice that brings a calm professional touch to corporate communications.\nAnaïs | ImBVnxSeLsdCfNIV | fr | fr | Young Adult | Feminine | A distinctive French young adult voice with a sharp tone perfect for lifestyle and character roles.\nMarine | GmGF_3ETsY2Zq7_w | fr | fr | Adult | Feminine | A warm and nurturing French adult voice ideal for storytelling education and empathetic support.\nMaxime | s0PhgjzOTRD5wo5L | fr | ca | Adult | Masculine | A joyful and instructional Canadian French voice that makes learning and support feel effortless.\nAlexandre | HBfu9XA3QfzAG1MN | fr | ca | Adult | Masculine | A high-energy and assertive Canadian French voice perfect for fast-paced promos and clear instructions.\nCamille | w9V1722uEmTkWqnR | fr | fr | Adult | Feminine | A joyful and professional French adult voice that delivers corporate and journalistic scripts with energy.\nMarie | BbLb4TxdlrldgpHI | fr | fr | Adult | Feminine | A warm and professional French adult voice ideal for calm instruction and empathetic communication.\nThomas | 8nsAoui8Y5RK9PYw | fr | fr | Adult | Masculine | A confident and sincere French adult voice that drives action in commercials and educational explainers.\nChloé | rIYDMY3dLccdauWA | fr | fr | Adult | Feminine | A bright and versatile French adult voice perfect for friendly assistance education and lifestyle content.\nNicolas | mxcKXLymdLQCdlEq | fr | fr | Adult | Masculine | An assertive and warm French adult voice that brings strength and character to narration and promos.\nLaura | Jlh1B0PKQJyup0sQ | fr | fr | Adult | Feminine | A helpful and clear French adult voice that excels in both educational content and empathetic service.\nAmandine | NvHEAMGiPT4u8iT- | fr | fr | Adult | Feminine | A versatile and joyful French adult voice capable of shifting from warm education to playful character work.\nValentin | WWHSNJCSTm77dyGd | fr | fr | Adult | Masculine | A warm and lively French adult voice that brings a spark of genuine enthusiasm to any script.\nManu | L6OaiBybqikfCBk0 | fr | fr | Young Adult | Masculine | A pleasant voice with a low pitch and smooth texture.\nSofia | s4CzgVHP5cEkB9LD | es | es | Adult | Feminine | Soft low-pitched and smooth with a slow and measured pace.\nPablo | aCWBiYUiQ4VwW8_b | es | es | Adult | Masculine | A warm low-pitched Spanish adult voice that brings a calm smooth authority to any script.\nCarlos | yPxeHKlCzaHeKd_V | es | es | Adult | Masculine | A warm and versatile Spanish adult voice that adapts seamlessly from ads to professional settings.\nAdrián | r5WB0b126tlHSrku | es | mx | Young Adult | Masculine | A warm and smooth Mexican young adult voice that naturally bridges journalism and conversation.\nAlberto | h39kz1iyoymcjcqh | es | es | Young Adult | Masculine | A warm Spanish young adult voice with a hosting flair perfect for media and customer engagement.\nElena | PqjKPYFyGNsg1YU- | es | es | Young Adult | Feminine | A warm and engaging Spanish young adult voice that makes journalism and education feel accessible.\nJavier | wGhY_zZCoQ5gB0ce | es | ar | Adult | Masculine | A warm and smooth Argentine adult voice that delivers professional and social content with charm.\nSergio | -8ZoUJpVU98rxpv9 | es | mx | Young Adult | Masculine | An energetic Mexican young adult voice that brings a bright modern feel to customer service and ads.\nDavid | zdE2H9vw2vcMl_Pt | es | mx | Adult | Masculine | A joyful and smooth Mexican adult voice that fits perfectly in both casual chats and formal spots.\nAna | ynR4CAbXMiOv-vGC | es | es | Young Adult | Feminine | A warm and versatile Spanish young adult voice ideal for everything from ads to professional service.\nSara | lPCVUcicz2XRaLE3 | es | es | Adult | Feminine | A warm and knowledgeable Spanish adult voice that balances journalistic clarity with conversational ease.\nMarta | VAb2M8nKHlUUZBk4 | es | mx | Young Adult | Feminine | A warm and relatable Mexican young adult voice perfect for connecting with Gen Z audiences.\nDaniel | R3L8t75ZEoZCPUA9 | es | es | Adult | Masculine | A confident low-pitched Spanish adult voice that commands respect in professional and service contexts.\nAlejandro | eorxD0DWv--n7l3p | es | es | Young Adult | Masculine | A joyful and smooth Spanish young adult voice that adds a fresh energy to advertisements.\nCristina | Bwl2KLUPxf82_ZaJ | es | mx | Adult | Feminine | A joyful and resonant Mexican adult voice ideal for vibrant social media and character work.\nCarmen | zhH3lPUo-JxmlOJT | es | co | Young Adult | Feminine | An energetic Colombian young adult voice that captures the lively spirit of a millennial streamer.\nMaría | k2B3TJiffePxjeBn | es | co | Young Adult | Feminine | A warm and smooth Colombian young adult voice that brings a friendly touch to education and ads.\nMiguel | Gijj_GPBfJVcP-FZ | es | es | Adult | Masculine | A steady Spanish adult voice with a robotic edge perfect for automated customer service.\nLaura | xB86uC_i8sO2U41- | pt | br | Adult | Feminine | A smooth and pleasant voice perfect for a nice chat.\nFrederico | L7890s1B44FqSiGC | pt | br | Adult | Masculine | A clear low-pitched voice spoken with a smooth texture\nEduardo | hAdJ9w9xBQkFgrRl | pt | br | Adult | Masculine | A clear low-pitched voice with a smooth texture.\nRodrigo | EzmLkNorEpZG_oNv | pt | pt | Young Adult | Masculine | A low-pitched Portuguese young adult voice that delivers information with calm confidence.\nBruna | Du_Dcv4fgXBDdubR | pt | pt | Adult | Feminine | A high-pitched energetic Portuguese adult voice perfect for engaging corporate training and narration.\nDaniel | _cP-0vSYfMmzR4al | pt | br | Adult | Masculine | A joyful and dynamic Brazilian adult voice that brings excitement to radio hosting and promos.\nLeonardo | YUKEEk7Y4Igsj1Ts | pt | pt | Adult | Masculine | An energetic and varied Portuguese adult voice ideal for lively radio spots and character work.\nThiago | QZtWUy8jmIroWiOu | pt | br | Adult | Masculine | A warm and versatile Brazilian adult voice that balances professional hosting with genuine kindness.\nPedro | Yee42wDKxEFHi0BS | pt | br | Young Adult | Masculine | A smooth low-pitched Brazilian young adult voice with a cool steady tone for scripts.\nMatheus | wT1bHy1Vq_0Bn73I | pt | pt | Adult | Masculine | A resonant and warm Portuguese adult voice that brings authority and kindness to educational content.\nJéssica | Fmt16x6anKfMMeSx | pt | br | Adult | Feminine | A smooth Brazilian adult voice designed for clear and professional customer service.\nFernando | 8QUaJGjSFdgHkuI8 | pt | br | Young Adult | Masculine | A warm and friendly Brazilian young adult voice that sounds like the approachable guy next door.\nJuliana | B6aHVROMF8FuKR07 | pt | pt | Young Adult | Feminine | A high-pitched energetic Portuguese young adult voice perfect for animated characters and lively dialogue.\nAna | 24cfpJbYGXZLE39T | pt | br | Adult | Feminine | A joyful and neighborly Brazilian adult voice that feels instantly familiar and welcoming.\nGustavo | T4yRIRCLji61Fz-N | pt | br | Adult | Masculine | A high-pitched friendly Brazilian adult voice ideal for approachable and caring roles.\nBruno | isyT17KHEj84P9w9 | pt | br | Adult | Masculine | A warm and helpful Brazilian adult voice that conveys genuine reliability and kindness.\nMaria | 73lMH7Zcc411nxJz | pt | pt | Adult | Feminine | A cheerful and helpful Portuguese adult voice that brightens any conversational script.\nLetícia | h6qFHXR3-bqPg_PE | pt | br | Adult | Feminine | A warm and empathetic Brazilian adult voice perfect for podcasting and supportive messaging.\nRafael | KpDAXeGeen7P9Uri | pt | pt | Adult | Masculine | A warm and friendly Portuguese adult voice ideal for relatable radio hosting and conversation.\nGabriel | 4ubKCfFxLeBg-cbl | pt | br | Adult | Masculine | An energetic and joyful Brazilian adult voice that commands attention with charismatic flair.\nLucas | AaTW_13X1yYe_OnX | pt | br | Adult | Masculine | A warm low-pitched Brazilian adult voice that adds a kind educational tone to any project.\nJoão | YHOBjtajNBEHUI_K | pt | br | Adult | Masculine | A smooth and clear Brazilian adult voice perfect for conversational delivery.\nMoritz | IIZIkBSZAmb9nFZb | de | de | Adult | Masculine | Clear low-pitched male voice with a smooth texture and a slow measured pace.\nLisa | kAoOc9Yb5EQDzA-N | de | de | Adult | Feminine | A soft and clear voice with a varied pitch.\nHans | vbg20SqFS_gBntTQ | de | at | Adult | Masculine | A calm low-pitched male delivery with a pleasant tone.\nFranziska | VXA4-0_ZN4o8q3vK | de | de | Adult | Feminine | A warm and smooth German adult voice that offers deep support with kindness.\nDavid | zyla-_bhVQtNTBdT | de | de | Adult | Masculine | A smooth German adult voice that educates with a calm low tone.\nLea | lSVEPWl_N_7MtcHe | de | de | Adult | Feminine | A warm and smooth German adult voice that teaches with a friendly approachable style.\nStefanie | hXjVvZ6oDDGQAQFj | de | de | Young Adult | Feminine | A confident and airy German young adult voice that reads with sincerity and clarity.\nTom | xq0vDziADfAmg6Uh | de | de | Adult | Masculine | An airy German adult voice that speaks publicly with a formal high pitch.\nNiklas | -qKylkN2UPxd7Mmg | de | de | Adult | Masculine | A joyful and smooth German adult voice that handles customer care with formal positivity.\nMichelle | fJDF4lEH590XplFv | de | de | Adult | Feminine | A joyful and smooth German adult voice that coaches with high energy and encouragement.\nJasmin | h2o5CDDhV5wE3Bwi | de | de | Adult | Feminine | A balanced and airy German adult voice that makes book reading feel light and accessible.\nDominik | ZOiGbnYdgKSBM_rH | de | de | Adult | Masculine | A balanced and smooth German adult voice designed for formal customer care.\nSabrina | dK5Glio51HTxdMu0 | de | de | Adult | Feminine | A balanced and smooth German adult voice perfect for professional book reading.\nDennis | YHkMHL6WppbXd42a | de | de | Adult | Masculine | An airy German adult voice that delivers technical information with formal grace.\nJulian | LAmPTQZkwYJKRCKt | de | de | Adult | Masculine | A balanced and resonant German adult voice that coaches with a calm steady presence.\nJannik | RPw-aWdY8NBiIWeg | de | de | Adult | Masculine | A joyful and resonant German adult voice that motivates and coaches with authority.\nMelanie | bauuigqCZbJFfk5q | de | de | Adult | Feminine | A warm and smooth German adult voice that brings a professional deep perspective.\nChristian | WxHB2b5HxA0Kuq5u | de | de | Adult | Masculine | A joyful and smooth German adult voice that reports with energy and professionalism.\nNadine | 9O8ZawShJ7UwURjK | de | de | Adult | Feminine | A warm and smooth German adult voice that educates with journalistic precision.\nNicole | t1Y_yKjku5R46F9t | de | de | Adult | Feminine | A warm and airy German adult voice that delivers journalistic content with a kind touch.\nSebastian | KEMqb7dQlTCAEUx6 | de | de | Mature | Masculine | A steady resonant German mature voice that brings the comforting wisdom of a grandfather.\nLena | df4Al5gt14Am4Qaf | de | de | Adult | Feminine | A grounded and smooth German adult voice that reports the news with a steady tone.\nFabian | 42-EbMFThYfhVB83 | de | de | Adult | Masculine | A warm German adult voice that teaches with a high engaging energy.\nPatrick | 3-pqEMoGtIq7wXtH | de | de | Adult | Masculine | A steady and smooth German adult voice that explains educational topics with journalistic clarity.\nChristina | 9LhjfdN9LOrygqDi | de | de | Adult | Feminine | A warm and smooth German adult voice that offers insightful guidance with a friendly tone.\nJessica | --9DFXOPx8kJFsbe | de | de | Adult | Feminine | An airy German adult voice that delivers formal journalism with a light touch.\nJennifer | XFttJvHwReWtWQNQ | de | de | Adult | Feminine | A steady and smooth German adult voice that reports professionally and formally.\nVanessa | 8eZwfGLoSF2N0RB3 | de | de | Adult | Feminine | A warm and smooth German adult voice that engages listeners as a lively podcast host.\nMaria | sz-H9BxaRaqxQ2S0 | de | de | Adult | Feminine | A relaxed and airy German adult voice that hosts podcasts with a cool vibe.\nJonas | 6tFmjkrmrdhO2bXV | de | de | Adult | Masculine | A warm German adult voice that contemplates and converses with philosophical insight.\nAnna | D8iRHK1qJhqfE00v | de | de | Adult | Feminine | A balanced airy German adult voice that brings deep empathy to conversation.\nMarcel | Cw79FL0p0J6UM9El | de | de | Adult | Masculine | A balanced German adult voice that handles customer care with a clear high-pitched tone.\nKevin | AySdCEnP2nqRo1WM | de | de | Adult | Masculine | A steady and smooth German adult voice that maintains a formal journalistic standard.\nTobias | uycTGmIXbw_Y83p9 | de | de | Adult | Masculine | A low-pitched German adult voice that delivers technical details with care and precision.\nDaniel | H0GE4TqfCQGmpQhL | de | de | Adult | Masculine | A warm and resonant German adult voice that sounds like a friendly student peer.\nTim | -WFy9WtlQNE-dEV2 | de | de | Adult | Masculine | A steady and resonant German adult voice ideal for professional customer care interactions.\nPhilipp | ZsVFAOnjnEPxJVDI | de | de | Adult | Masculine | A balanced German adult voice that reads books with a smooth immersive flow.\nMaximilian | H3Rh9kJcd4gZidvN | de | de | Adult | Masculine | A warm German adult voice that educates with a calm low-pitched authority.\nSarah | ApPgTz3nMHOsWxhK | de | de | Adult | Feminine | A warm low-pitched German adult voice that offers the soothing understanding of a close confidant.\nFlorian | XnSnbQW98he4aULg | de | de | Adult | Masculine | A warm German adult voice that delivers news with a high resonant clarity.\nKatharina | AEJ61XaIaRill4cJ | de | de | Adult | Feminine | A steady low-pitched German adult voice designed for steady and engaging book reading.\nMona | T2NDxsof9FHYxgJj | de | de | Adult | Feminine | A warm and smooth German adult voice that brings a tutor's patience to any script.\nLaura | wBgI9XmASQwvQ13w | de | de | Adult | Feminine | A warm German adult voice that teaches and guides with a kind high-pitched tone.\nFelix | uF8PfAXrv6qU9UEM | de | de | Adult | Masculine | A smooth German adult voice perfect for straightforward journalistic reporting.\nJulia | FRTqjB2TL-Ix9GXW | de | de | Adult | Feminine | A warm and conversational German adult voice that sounds like a relatable student.\nAlexander | xki1DK6Ks6tuDmcb | de | de | Adult | Masculine | A warm German adult voice that reports with journalistic integrity and a resonant tone.\nLukas | 5UkFVe2B8OqLo-5R | de | de | Adult | Masculine | A low-pitched German adult voice that conveys the authority of a seasoned expert.\nJan | 1D38wv1wp-H7QcyM | de | de | Adult | Masculine | A balanced German adult voice with a high pitch ideal for clear customer care.\n \n\n
\n\n# Custom Voices\n\nCreate and manage your own custom voice clones. Custom voices are passed to TTS using the `voice_id` parameter (not `voice`).\n\n## List All Custom Voices\n\n```python\nimport json\nimport gradium\n\nall_custom_voices = await gradium.voices.get(client)\nprint(json.dumps(all_custom_voices, indent=2))\n```\n\n## Get Specific Voice\n\n```python\nimport json\n\nvoice = await gradium.voices.get(client, voice_uid=\"abc123def456\")\nprint(json.dumps(voice, indent=2))\n```\n\n## Create Custom Voice\n\n```python\nimport json\n\nvoice = await gradium.voices.create(\n client,\n audio_file=\"my_voice_sample.wav\",\n name=\"My Custom Voice\",\n description=\"A voice created from my recording\",\n start_s=0.0,\n)\nprint(json.dumps(voice, indent=2))\n```\n\n## Update Voice\n\n```python\nawait gradium.voices.update(\n client,\n voice_uid=\"abc123def456\",\n name=\"Updated Voice Name\",\n description=\"Updated description\",\n start_s=1.5\n)\n```\n\n## Delete Voice\n\n```python\nawait gradium.voices.delete(client, voice_uid=\"abc123def456\")\n```\n\n# Credit Management\n\nCredits are consumed based on the audio generated: **1 credit equals 1 character of TTS**.\nOne minute is approximately 750 characters, so 1h of TTS generation is approximately 45 000 characters.\n\n## Get Credit Information\n\n```python\nimport json\n\ncredits_info = await gradium.usages.get(client)\nprint(json.dumps(credits_info, indent=2))\n```\n\n# Speech-to-Text (STT)\n\nThe Speech-to-Text model converts audio input into text transcriptions, supporting real-time streaming and a semantic VAD.\n\n## Basic Streaming Usage\n\n```python\nimport asyncio\nimport gradium\n\nasync def main():\n client = gradium.client.GradiumClient(api_key=\"your-api-key\")\n\n # Audio generator that yields audio chunks\n async def audio_generator(audio_data, chunk_size=1920):\n for i in range(0, len(audio_data), chunk_size):\n yield audio_data[i : i + chunk_size]\n\n # Create STT stream\n stream = await client.stt_stream(\n {\"model_name\": \"default\", \"input_format\": \"pcm\"},\n audio_generator(audio_data),\n )\n\n # Process transcription results\n async for message in stream.iter_text():\n print(message)\n\nif __name__ == \"__main__\":\n asyncio.run(main())\n```\n\n## Setup Parameters\n\n- **`model_name`**: The STT model to use (default: `\"default\"`)\n- **`input_format`**: Audio format of the input data (supported: `\"pcm\"`,\n `\"wav\"`, `\"opus\"`)\n\nWhen using `\"pcm\"` input format, the audio must adhere to the following\nspecifications:\n- **Sample Rate**: 24000 Hz (24kHz)\n- **Format**: PCM (Pulse Code Modulation)\n- **Bit Depth**: 16-bit signed integer\n- **Channels**: Single channel (mono)\n- **Chunk Size**: Recommended 1920 samples per chunk (80ms at 24kHz)\n\nWhen using `\"wav\"` input format, the audio must be a valid WAV file using\nPCM data (so `AudioFormat` = 1 in the WAV header). Supported bits per sample\nare 16, 24 and 32 bits.\n\nWhen using `\"opus\"` input format, the audio must be some ogg wrapped opus data\nstream.\n\n## Message Types\n\nThe STT stream returns different types of messages:\n- **Text Messages** (`text`): Contain transcription results together with timestamps.\n- **VAD Messages** (`step`): Provide Voice Activity Detection information to determine\n when the speaker has finished speaking.\n\n```python\n# Text messages containing transcription results\nasync for msg in stream._stream:\n if msg.get(\"type\") == \"text\":\n print(f\"Transcription: {msg}\")\n\n # VAD (Voice Activity Detection) messages\n elif msg.get(\"type\") == \"step\":\n vad_info = msg.get(\"vad\", {})\n # Use msg[\"vad\"][2][\"inactivity_prob\"] to detect turn completion\n # VAD steps occur every 80ms\n inactivity_probability = msg[\"vad\"][2].get(\"inactivity_prob\")\n print(f\"Inactivity probability: {inactivity_probability}\")\n```\n\n## Advanced Options\n\nSome models support advanced options that can be passed using the `json_config`\nparameter. In the Python api, this parameter is passed as a dictionary mapping\nstring to values (either float or string).\n\nThis parameter can be used to control:\n- Stability of the generated speech via the `text` temperature parameter.\n- Expected language via the `language` parameter.\n- Delay to generate the text in audio frames via the `delay_in_frames` parameter.\n\n**Temperature Control** Sets the temperature used for text generation. The\ndefault value is 0 resulting in some greedy sampling. Higher values (up to 1)\nresult in more diverse outputs, in particular these can be helpful if no\ntext is recognized.\n\n**Language Control** Sets the expected language of the audio. This can help\ngrounding the model to a specific language and improve transcription quality.\nIf multiple languages are expected, this can be set to the main language.\n\n**Delay Control** Sets the delay in audio frames (80ms each) before text\nis generated. Higher delays allow the model to gather more context before\ngenerating text, which can improve quality at the cost of latency.\nThe allowed values are `7, 8, 10, 12, 14, 16, 20, 24, 36, 48`.\n\n# Text Rewriting Rules\n\nThe text-to-speech API supports text rewriting rules that normalize and expand certain patterns in the input text before synthesis. These rules help the TTS model properly pronounce dates, times, numbers, email addresses, URLs, phone numbers, and alphanumeric codes.\n\n## Configuration\n\nRewrite rules can be enabled by adding a `rewrite_rules` field to the `json_config` in the setup message. The field accepts a comma-delimited string of rule names or language aliases.\n\n**Example setup message:**\n```json\n{\n \"json_config\": {\n \"rewrite_rules\": \"en\"\n }\n}\n```\n\nOr with specific rules:\n```json\n{\n \"json_config\": {\n \"rewrite_rules\": \"TimeEn,Date,NumberEn,EmailEn\"\n }\n}\n```\n\n## Language Aliases\n\nFor convenience, language aliases are provided that enable all recommended rules for a specific language:\n\n| Alias | Enabled Rules |\n|-------|---------------|\n| `en` | TimeEn, Date, AlNum, NumberEn, EmailEn, UrlEn, PhoneEn |\n| `fr` | TimeFr, Date, AlNum, NumberFr, EmailFr, UrlFr, PhoneFr |\n| `de` | TimeDe, Date, AlNum, NumberDe, EmailDe, UrlDe, PhoneDe |\n| `es` | Date, AlNum, NumberEs, EmailEs, UrlEs, PhoneEs |\n| `pt` | Date, AlNum, NumberPt, EmailPt, UrlPt, PhonePt |\n\n## Available Rewrite Rules\n\n### Date Rule\n\n**Rule name:** `Date`\n\nConverts numeric dates to a more speech-friendly format.\n\n**Examples:**\n- `12/31/2020` → `12-31 2020`\n- `16/01/1980` → `16-01 1980`\n- `1/5.` → `1-5.`\n\nThe rule preserves punctuation at the end of the date.\n\n### Time Rules\n\nTime rules convert various time formats to standardized representations for each language.\n\n#### TimeEn (English)\n\n**Rule name:** `TimeEn`\n\nConverts time formats with colons or periods, with optional AM/PM markers.\n\n**Examples:**\n- `3:45PM!` → `3.45PM!`\n- `12.30.` → `12.30.`\n- `12:30` → `12.30`\n\n#### TimeFr (French)\n\n**Rule name:** `TimeFr`\n\nConverts French time formats (with 'h' separator or colons).\n\n**Examples:**\n- `9h15,` → `9h15,`\n- `14:00?` → `14h00?`\n\n#### TimeDe (German)\n\n**Rule name:** `TimeDe`\n\nConverts German time formats (colons or periods).\n\n**Examples:**\n- `8:20.` → `8.20.`\n- `22.45!` → `22.45!`\n\n### Number Rules\n\nNumber rules expand large numbers into word-based representations for better pronunciation. Years (1900-2100) and small numbers (< 1000) are kept as-is.\n\n**Rule names:** `NumberEn`, `NumberFr`, `NumberDe`, `NumberEs`, `NumberPt`\n\n**English examples:**\n- `123` → `123` (small numbers unchanged)\n- `1234` → `1 thousand 234`\n- `1000000` → `1 million`\n- `2500000` → `2 million 500 thousand`\n- `1002003004` → `1 billion 2 million 3 thousand 4`\n- `-4500` → `minus 4 thousand 500`\n\n**French examples:**\n- `1234` → `mille 234` (singular form for 1)\n- `2234` → `2 mille 234`\n- `2000000` → `2 millions`\n- `-4500` → `moins 4 mille 500`\n- `123456000789` → `123 milliards 456 millions 789`\n\n**Language-specific separators:**\n- **English:** thousand, million, billion\n- **French:** mille, million(s), milliard(s)\n- **German:** Tausend, Million(en), Milliarde(n)\n- **Spanish:** mil, millón/millones, mil millones\n- **Portuguese:** mil, milhão/milhões, bilhão/bilhões\n\n### Email Rules\n\nEmail rules spell out email addresses with language-specific words for special characters.\n\n**Rule names:** `EmailEn`, `EmailFr`, `EmailDe`, `EmailEs`, `EmailPt`\n\n**English examples:**\n- `foo.bar@gmail.com` → `foo dot bar at gmail dot com`\n\n**French examples:**\n- `foo@gmail.com` → `foo arobaze gmail point com`\n\n**Special character translations:**\n- `@` → \"at\" (en), \"arobaze\" (fr), \"at\" (de), \"arroba\" (es), \"arroba\" (pt)\n- `.` → \"dot\" (en), \"point\" (fr), \"Punkt\" (de), \"punto\" (es), \"ponto\" (pt)\n- `-` → \"dash\" (en), \"tiret\" (fr), \"Bindestrich\" (de), \"guión\" (es), \"hífen\" (pt)\n\n### URL Rules\n\nURL rules spell out URLs including protocol, domain, path, and special characters.\n\n**Rule names:** `UrlEn`, `UrlFr`, `UrlDe`, `UrlEs`, `UrlPt`\n\n**English examples:**\n- `www.example.com` → `www dot example dot com`\n- `https://www.example.com/path` → `H-T-T-P-S colon slash slash www dot example dot com slash path`\n- `http://sub.domain.co.uk` → `H-T-T-P colon slash slash sub dot domain dot C-O dot U-K`\n\n**French examples:**\n- `https://www.kyutai.fr` → `H-T-T-P-S deux-points slash slash www point kyutai point F-R`\n- `www.it-management.com/promo` → `www point I-T tiret management point com slash promo`\n\nTwo-letter top-level domains are spelled out (e.g., \"UK\" → \"U-K\", \"FR\" → \"F-R\").\n\n### Phone Number Rules\n\nPhone number rules format phone numbers according to country-specific conventions.\n\n**Rule names:** `PhoneEn`, `PhoneFr`, `PhoneDe`, `PhoneEs`, `PhonePt`\n\nPhone numbers can be:\n- **International format:** Starting with `+` and a country code\n- **Local format:** Starting with `0`\n\n**French examples:**\n- `0123456789` → `01 23 45 67 89`\n- `+330556791936` → `+33 05 56 79 19 36` (French TTS)\n- `+330556791936` → `+33 0-5 5-6 7-9 1-9 3-6` (English TTS)\n\n**English examples:**\n- `07596854413` → `0-7-5-9 6-8-5 4-4-1-3`\n- `+16502349653` → `+1 6-5-0 2-3-4 9-6-5-3`\n- `+447700900123` → `+44 7-7-0 0-9-0 0-1-2-3` (mobile)\n- `+442000900123` → `+44 2-0 0-0-9-0 0-1-2-3` (London)\n\n**German examples:**\n- `01511234567` → `0-1-5 1 1 2-3 4-5 6-7`\n- `+491511234567` → `+49 1-5 1 1 2-3 4-5 6-7`\n\n**Supported country codes:**\n- `+1` - North America\n- `+33` - France\n- `+34` - Spain\n- `+44` - United Kingdom\n- `+49` - Germany\n- `+351` - Portugal\n\n### AlNum (Alphanumeric)\n\n**Rule name:** `AlNum`\n\nHandles mixed uppercase letters and digits (e.g., license plates, product codes).\n\n**Examples:**\n- `AB12CD34!` → `A-B 1-2 C-D 3-4!`\n\nCharacters are grouped by type (letters vs. digits) and joined with hyphens within each group.\n\n## Best Practices\n\n1. **Use language aliases** when possible for comprehensive coverage in a single language\n2. **Combine specific rules** when you need fine-grained control or multi-language support\n3. **Preserve punctuation** - rules preserve trailing punctuation (periods, commas, etc.)\n4. **International phone numbers** require at least 6 digits to be recognized\n5. **Year detection** - numbers between 1900-2100 are kept as-is and not expanded\n\n## Implementation Notes\n\n- Rules are applied word-by-word to the input text\n- Only the first matching rule is applied to each word\n- Special characters like quotes, dashes, and brackets are normalized before processing\n- Colons (`:`) are handled specially to support time and URL formats\n- When no rules are specified, minimal text normalization is applied\n", "x-displayName": "Documentation" }, { "name": "FAQ", "description": "Frequently Asked Questions\n=======================\n\n__What language do you support?__\n\nWe currently support English, French, Spanish, Portuguese, and German, with more\nlanguages currently in development. Sign up to be updated when more languages\nbecome available!\n\n__What is the maximum session duration?__\n\nA session can last up to 300 seconds. If you want to generate longer chunks of\ntext or transcribe longer audio, it's better to split it into different\nsessions.\n\nWhen using the free tier, there is an additional limitation of 1500 characters\nper session.\n\n__How many credits do I need?__\n\nFor Text-to-Speech, one minute of audio typically requires 750 characters,\nwhich corresponds to 750 credits. This may vary depending on the language and\nthe speaker’s pace. For Speech-to-Text, each second of audio costs 3 credits\n\n__Can I use my own voice?__\n\nYes! We offer two levels of voice cloning. Our standard feature allows you to\ncreate a high-quality clone with just a 10-second audio sample. For those\nseeking even higher fidelity, we now offer Pro Voice Clones; by providing 30\nminutes to a few hours of audio data, you can create a custom voice with\nunmatched quality and nuance.\n\n__What is an Instant Voice Clone?__\n\nInstant voice cloning enables you to create realistic digital replicas using just\na few seconds of reference audio (typically 10 seconds). Depending on your\nsubscription plan, you can clone up to 1,000 voices. Please note that explicit\nconsent from the voice owner is required.\n\n__What is a Pro Voice Clone?__\n\nA Pro Voice Clone is a high-fidelity, hyper-realistic voice model created by\nfine-tuning our dedicated AI on a large dataset of your audio. Unlike standard\ncloning, this process captures the speaker's deepest emotional nuances, unique\naccents, and natural pacing, resulting in a digital voice indistinguishable from\nthe original.\n\n__How do I generate a Pro Voice Clone?__\n\nTo get started, navigate to the Pro Voice Clone tab in Gradium Studio and upload\nyour audio dataset. You will receive a notification once the upload is processed. After\ntraining is complete, the voice will appear in your library and be ready for\nText-to-Speech (TTS) generation.\n\n__How much audio do I need for a voice clone?__\n\nFor an Instant Voice Clone, we get optimal results with only 10 seconds of data. For a\nPro Voice Clone, we require a minimum of 30 minutes of clean audio data. For optimal results\nwhere the voice captures full emotional range and stability, we recommend\nproviding 2 hours of audio.\n\n" }, { "name": "Release notes", "description": "## 2026.02\n\nWe're excited to announce major updates to the Gradium API and Gradium Studio, delivering enhanced text-to-speech (TTS) and speech-to-text (STT) capabilities for developers and enterprises.\n\n🎙️ **Advanced TTS and STT Models**\n\nExperience our latest text-to-speech model and speech recognition model with improved audio quality, accuracy, and natural-sounding voice generation. Perfect for voice applications, transcription services, and conversational AI. Now by default \n\n📖 **Custom Pronunciation Dictionaries**\n\nTake control of speech synthesis with pronunciation dictionaries. Ensure brand names, technical terminology, and industry-specific acronyms are pronounced correctly every time. Ideal for healthcare, finance, and industry specific applications.\n\n⚡ **WebSocket Multiplexing for Real-Time Audio**\n\nBoost performance with multiplexing support: process multiple TTS or STT requests simultaneously over a single WebSocket connection. Reduce latency, minimize connection overhead, and scale efficiently for high-volume applications.\n\n🎁 **Gradium Referral Program**\n\nLove Gradium? Join our referral program: share Gradium with your network and earn up to 9M API credits while your referrals get exclusive discounts. Win-win for developers and businesses alike.\n\n☁️ **AWS Marketplace & SageMaker AI Deployment**\n\nDeploy Gradium TTS and ASR models directly on AWS SageMaker with full bidirectional streaming support. Accelerate your deployment timeline while keeping your voice AI infrastructure secure within your private cloud network. Now available on AWS Marketplace.\n\nFeedbacks are always welcome, do not hesitate to join our [discord channel](https://discord.com/invite/bcysuPRzXE)!\n" }, { "name": "TTS", "description": "Text-to-Speech endpoints for converting text to audio" }, { "name": "STT", "description": "Speech-to-Text endpoints for converting audio to text" }, { "name": "Voices", "description": "Manage custom voice clones" }, { "name": "Pronunciations", "description": "Manage pronunciation dictionaries for custom text rewriting" }, { "name": "Credits", "description": "Monitor API credit balance" } ] }