{ "schema": "chi-bench/submission/v1", "submission": { "id": "glm-5-2-openai-agents", "team": "Actava", "contact": "dark.savi@gmail.com", "agent": "openai-agents", "model": "z-ai/glm-5.2", "notes": "GLM-5.2 via the OpenAI Agents SDK harness, routed through OpenRouter.\n", "submitted_at": "2026-07-06T05:23:05Z" }, "dataset": { "version": "chi-bench-v1.0.0", "domains": [ "pa_provider", "pa_um", "cm" ], "name": "chi-bench" }, "results": { "overall": { "n_trials": 75, "n_tasks": 75, "pass_at_1": 0.18666666666666668, "mean_cost_usd": 0.15959464888, "mean_walltime_s": 0.0 }, "per_domain": { "pa_provider": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.2, "mean_cost_usd": 0.15565148484, "mean_walltime_s": 0.0 }, "pa_um": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.32, "mean_cost_usd": 0.1808380908, "mean_walltime_s": 0.0 }, "cm": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.04, "mean_cost_usd": 0.142294371, "mean_walltime_s": 0.0 } }, "mean_cost_usd": 0.15959464888, "mean_walltime_s": 0.0 }, "provenance": { "chi_bench_git_sha": "ee1e16d60afd8bc59a6d90a480fede19593ea84f", "image_digest": null, "judge_model": "claude-opus-4-7", "harness_version": "0.1.0", "code_dirty": true, "dataset_version": "chi-bench-v1.0.0", "environment": "modal", "started_at": "2026-07-02T04:22:04Z", "finished_at": "2026-07-02T04:48:05Z", "host": "Haolins-MacBook-Pro.local", "python_version": "3.12.12", "platform": "macOS-26.5.1-arm64-arm-64bit" } }