{ "model_name": "gemini-3.1-flash-live-preview-thinking-high", "model_organization": "Google", "submitting_organization": "Sierra", "submission_date": "2026-07-13", "submission_type": "custom", "modality": "voice", "contact_info": { "email": "soham@sierra.ai", "name": "Sierra Research Team" }, "results": { "banking_knowledge": { "pass_1": 11.34020618556701, "retrieval_config": "alltools" } }, "is_new": true, "trajectories_available": true, "trajectory_files": { "banking_knowledge": "gemini-3-1-flash-live-preview_voice_high_usersim-gpt55xhigh_banking" }, "methodology": { "evaluation_date": "2026-07-13", "tau2_bench_version": "1.0.1", "user_simulator": "gpt-5.5-2026-04-23 (xhigh)", "notes": "Custom submission: first voice-banking (voice + knowledge) entry. Identical to the standard voice evaluation pipeline (audio-native, regular speech complexity, ElevenLabs eleven_v3 TTS, Gemini thinking level HIGH) but run on the banking_knowledge domain with retrieval-config=alltools, and the user simulator LLM is gpt-5.5-2026-04-23 at reasoning_effort=xhigh instead of the leaderboard-standard v1.0 (gpt-4.1). Not comparable to standard-user-simulator leaderboard entries. Re-evaluated under tau2-bench v1.0.1 (banking_knowledge grading/task fixes #329/#388/#397/#402/#403): all 97 trajectories re-graded with the fixed evaluate-trajs path, and the 27 calls whose agents were exposed to the pre-fix environment (transaction ordering, Platinum Rewards doc title) re-run live under the fixed environment regardless of prior outcome (2 gained, 4 previously-passing re-runs lost to run variance).", "verification": { "modified_prompts": false, "omitted_questions": false } }, "voice_config": { "provider": "gemini", "model": "gemini-3.1-flash-live-preview", "tick_duration_seconds": 0.2, "max_steps_seconds": 1200.0, "user_tts_provider": "elevenlabs/eleven_v3" }, "model_release": { "release_date": "2026-03-26", "announcement_url": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-1-flash-live/", "announcement_title": "Gemini 3.1 Flash Live: Google's latest AI audio model" }, "reasoning_effort": "high", "interaction_metrics": { "version": "1.0", "config": { "tick_duration_sec": 0.2, "no_yield_window_sec": 2.0, "backchannel_yield_window_sec": 1.0, "vocal_tic_yield_window_sec": 1.0, "non_directed_yield_window_sec": 1.0, "vocal_tic_response_window_sec": 2.0, "non_directed_response_window_sec": 2.0 }, "domains": { "banking_knowledge": { "response_latency_mean": 2.9520072992700728, "yield_latency_mean": 0.8205607476635515, "response_rate": 0.822205551387847, "yield_rate": 0.4942263279445728, "agent_interruption_rate": 0.17704426106526633, "selectivity_backchannel": 0.9344262295081968, "selectivity_vocal_tic": 0.5915492957746479, "selectivity_non_directed": 0.4387755102040817, "counts": { "n_simulations": 97, "response_total": 1333, "yield_total": 866, "backchannel_total": 61, "vocal_tic_total": 142, "non_directed_total": 98, "agent_interrupts_count": 236 } } }, "overall": { "response_latency_mean": 2.9520072992700728, "yield_latency_mean": 0.8205607476635515, "response_rate": 0.822205551387847, "yield_rate": 0.4942263279445728, "agent_interruption_rate": 0.17704426106526633, "selectivity_backchannel": 0.9344262295081968, "selectivity_vocal_tic": 0.5915492957746479, "selectivity_non_directed": 0.4387755102040817, "counts": { "agent_interrupts_count": 236, "backchannel_total": 61, "n_simulations": 97, "non_directed_total": 98, "response_total": 1333, "vocal_tic_total": 142, "yield_total": 866 } } } }