{ "schema": "chi-bench/submission/v1", "submission": { "id": "inkling-256k-openai-agents", "team": "Actava", "contact": "dark.savi@gmail.com", "agent": "openai-agents", "model": "thinkingmachines/Inkling:peft:262144", "notes": "Inkling 256K via the OpenAI Agents SDK harness and Tinker Chat Completions, with high reasoning effort and a 50-turn limit. Reported cost reflects recorded token usage; 25 terminal agent failures reported zero usage and therefore contribute $0 to the aggregate cost.\n", "submitted_at": "2026-07-24T23:02:07Z" }, "dataset": { "version": "chi-bench-v1.0.0", "domains": [ "pa_provider", "pa_um", "cm" ], "name": "chi-bench" }, "results": { "overall": { "n_trials": 75, "n_tasks": 75, "pass_at_1": 0.08, "mean_cost_usd": 0.9234470439466665, "mean_walltime_s": 140.97706868 }, "per_domain": { "pa_provider": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.04, "mean_cost_usd": 0.8389722792, "mean_walltime_s": 121.97677540000001 }, "pa_um": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.16, "mean_cost_usd": 1.4461965891200002, "mean_walltime_s": 213.6378354 }, "cm": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.04, "mean_cost_usd": 0.48517226352000004, "mean_walltime_s": 87.31659524000001 } }, "mean_cost_usd": 0.9234470439466665, "mean_walltime_s": 140.97706868 }, "provenance": { "chi_bench_git_sha": "4f39c1e77a90c28cd28e3f2264a8389a0dede039", "image_digest": null, "judge_model": "claude-opus-4-7", "harness_version": "0.1.0", "code_dirty": true, "dataset_version": "chi-bench-v1.0.0", "environment": "modal", "started_at": "2026-07-22T14:25:50.976659Z", "finished_at": "2026-07-22T16:53:02.466832Z", "source": "frontier_models_full_2026_07 experiment matrix", "trajectory_repair": { "count": 25, "reason": "Terminal agent failures completed verification but emitted no trace.jsonl.", "method": "Generated ATIF-v1.2 error-only trajectories with the official openai_agents_harness._build_atif_trajectory(records=[], ...) builder from each preserved agent/run_result.json; original logs were not modified." }, "recorded_cost_caveat": "The 25 terminal failures with error-only trajectories reported zero input/output/cache tokens and null cost; leaderboard cost is therefore a lower bound on billed execution cost." } }