{ "model_name": "xai-realtime", "model_organization": "xAI", "submitting_organization": "Sierra", "submission_date": "2026-07-13", "submission_type": "custom", "modality": "voice", "contact_info": { "email": "soham@sierra.ai", "name": "Sierra Research Team" }, "results": { "banking_knowledge": { "pass_1": 16.49484536082474, "retrieval_config": "alltools" } }, "is_new": true, "trajectories_available": true, "trajectory_files": { "banking_knowledge": "xai-realtime_voice_usersim-gpt55xhigh_banking" }, "methodology": { "evaluation_date": "2026-07-13", "tau2_bench_version": "1.0.1", "user_simulator": "gpt-5.5-2026-04-23 (xhigh)", "notes": "Combined voice+knowledge track: banking_knowledge domain run through the audio-native voice pipeline (regular speech complexity, alltools retrieval). Custom user simulator gpt-5.5-2026-04-23 (reasoning_effort=xhigh). Evaluated via xAI's 'xai-realtime' endpoint, which does not disclose the served Grok voice model version; Released is recorded from xAI's Grok Voice Think Fast 1.0 announcement (2026-04-23), the voice model line behind the endpoint. The endpoint exposes no reasoning-effort level, so Reasoning is recorded as enabled without a level. 1 trial per task. Re-evaluated under tau2-bench v1.0.1 (banking_knowledge grading/task fixes #329/#388/#397/#402/#403): all 97 trajectories re-graded with the fixed evaluate-trajs path (zero flips), and the 34 calls whose agents were exposed to the pre-fix environment (transaction ordering, Platinum Rewards doc title) re-run live under the fixed environment regardless of prior outcome (+6 passes: task_001/023/025/057 via the doc fix, task_093/095 via the ordering fix; -1: previously-passing task_058 lost on re-run).", "verification": { "modified_prompts": false, "omitted_questions": false } }, "voice_config": { "provider": "xai", "model": "xai-realtime", "tick_duration_seconds": 0.2, "max_steps_seconds": 1200.0, "user_tts_provider": "elevenlabs/eleven_v3" }, "interaction_metrics": { "version": "1.0", "config": { "tick_duration_sec": 0.2, "no_yield_window_sec": 2.0, "backchannel_yield_window_sec": 1.0, "vocal_tic_yield_window_sec": 1.0, "non_directed_yield_window_sec": 1.0, "vocal_tic_response_window_sec": 2.0, "non_directed_response_window_sec": 2.0 }, "domains": { "banking_knowledge": { "response_latency_mean": 1.4283085633404091, "yield_latency_mean": 0.985191637630662, "response_rate": 0.9978813559322034, "yield_rate": 0.9526970954356846, "agent_interruption_rate": 0.25, "selectivity_backchannel": 0.6818181818181819, "selectivity_vocal_tic": 0.39655172413793105, "selectivity_non_directed": 0.15503875968992253, "counts": { "n_simulations": 97, "response_total": 1416, "yield_total": 1205, "backchannel_total": 88, "vocal_tic_total": 116, "non_directed_total": 129, "agent_interrupts_count": 354 } } }, "overall": { "response_latency_mean": 1.4283085633404091, "yield_latency_mean": 0.985191637630662, "response_rate": 0.9978813559322034, "yield_rate": 0.9526970954356846, "agent_interruption_rate": 0.25, "selectivity_backchannel": 0.6818181818181819, "selectivity_vocal_tic": 0.39655172413793105, "selectivity_non_directed": 0.15503875968992253, "counts": { "agent_interrupts_count": 354, "backchannel_total": 88, "n_simulations": 97, "non_directed_total": 129, "response_total": 1416, "vocal_tic_total": 116, "yield_total": 1205 } } }, "reasoning_effort": "enabled", "model_release": { "release_date": "2026-04-23", "announcement_url": "https://x.ai/news/grok-voice-think-fast-1", "announcement_title": "Grok Voice Think Fast 1.0" } }