{ "schema": "chi-bench/submission/v1", "submission": { "id": "cuilinke-hermes-medguard", "team": "cuilinke", "contact": "13856240901@163.com", "agent": "hermes", "model": "MedGuard", "notes": "Local Hermes harness evaluation for MedGuard against an OpenAI-compatible model endpoint\nconfigured in .env.MedGuard via OPENAI_API_KEY and OPENAI_BASE_URL.\n", "submitted_at": "2026-07-06T09:10:18Z" }, "dataset": { "version": "chi-bench-v1.0.0", "domains": [ "pa_provider", "pa_um", "cm" ], "name": "chi-bench" }, "results": { "overall": { "n_trials": 75, "n_tasks": 75, "pass_at_1": 0.22666666666666666, "mean_cost_usd": 0.0, "mean_walltime_s": 0.0 }, "per_domain": { "pa_provider": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.04, "mean_cost_usd": 0.0, "mean_walltime_s": 0.0 }, "pa_um": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.04, "mean_cost_usd": 0.0, "mean_walltime_s": 0.0 }, "cm": { "n_trials": 25, "n_tasks": 25, "pass_at_1": 0.6, "mean_cost_usd": 0.0, "mean_walltime_s": 0.0 } }, "mean_cost_usd": 0.0, "mean_walltime_s": 0.0 }, "provenance": { "chi_bench_git_sha": "ac7de9f919fd3e9b69f4f521ae1e68607d23e995", "image_digest": "sha256:7494575a1cd12523366de99847e5303c8a8f302d8c5eeb74ccef3cc9513aa46b", "judge_model": "claude-opus-4-7", "harness_version": "0.1.0", "code_dirty": true, "dataset_version": "chi-bench-v1.0.0", "environment": "docker", "started_at": "2026-07-01T06:59:37Z", "finished_at": "2026-07-01T08:17:44Z", "host": "di-20260319190226-bjxhm", "python_version": "3.13.13", "platform": "Linux-5.4.250-2-velinux1u1-amd64-x86_64-with-glibc2.35" } }