{ "schema": "chi-bench/submission/v1", "submission": { "id": "chi-bench-leaderboard-2026-05-15-claude-code-anthropic-claude-opus-4-6", "team": "chi-Bench Leaderboard", "contact": "leaderboard@actava.ai", "agent": "claude-code", "model": "anthropic/claude-opus-4-6", "notes": "Paper-baseline submission re-staged from the existing leaderboard example under the unified chi-bench-leaderboard--- id.\n", "submitted_at": "2026-05-12T22:39:33Z" }, "dataset": { "version": "chi-bench-v1.0.0", "domains": [ "pa_provider", "pa_um", "cm" ], "name": "chi-bench" }, "results": { "overall": { "n_trials": 150, "n_tasks": 75, "pass_at_1": 0.28, "mean_cost_usd": 1.0094868733333333, "mean_walltime_s": 0.0 }, "per_domain": { "pa_provider": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.2, "mean_cost_usd": 0.8373129, "mean_walltime_s": 0.0 }, "pa_um": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.36, "mean_cost_usd": 1.00361896, "mean_walltime_s": 0.0 }, "cm": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.28, "mean_cost_usd": 1.18752876, "mean_walltime_s": 0.0 } }, "mean_cost_usd": 1.0094868733333333, "mean_walltime_s": 0.0 }, "provenance": { "chi_bench_git_sha": "f926f8f47a748872644646416d2e504d079da7ff", "image_digest": null, "judge_model": "claude-opus-4-7", "harness_version": "0.1.0", "code_dirty": true, "dataset_version": "chi-bench-v1.0.0", "environment": "modal", "started_at": "2026-05-12T18:55:45Z", "finished_at": "2026-05-12T18:55:46Z", "host": "Weirans-MacBook-Pro.local", "python_version": "3.13.12", "platform": "macOS-15.5-arm64-arm-64bit-Mach-O" } }