name: behavioral eval # Runs `claude plugin eval` against the plugin's evals/ suite: each case is a # prompt run in an isolated session with only this plugin loaded, scored by its # graders, plus a no-plugin baseline arm for the score delta. # # Push-to-main only (plus manual dispatch): the run spends the maintainer's # Claude subscription quota via CLAUDE_CODE_OAUTH_TOKEN (`claude setup-token`), # so it must never trigger on fork PRs. CLAUDE_CODE_WALNUT_SPIRE enables the # early-access `plugin eval` command on machines that cannot receive the # per-organization rollout — CI runners are its documented audience. on: push: branches: [main] paths: - skills/** - evals/** - plugin.json - .claude-plugin/** - .github/workflows/behavioral-eval.yml workflow_dispatch: permissions: contents: read jobs: eval: runs-on: ubuntu-latest timeout-minutes: 30 env: CLAUDE_CODE_OAUTH_TOKEN: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} CLAUDE_CODE_WALNUT_SPIRE: "1" steps: - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: install claude code run: | curl -fsSL https://claude.ai/install.sh | bash echo "$HOME/.local/bin" >> "$GITHUB_PATH" - name: materialize claude.ai credentials # The eval sandbox authenticates each run's child by copying the # parent's credentials file (~/.claude/.credentials.json on Linux) # into the run's fresh config dir. CLAUDE_CODE_OAUTH_TOKEN is not on # the sandbox's env pass-through list (only ANTHROPIC_API_KEY-style # env auth is), so on a runner that never ran `claude login` the # child was rejected at the first run (partialReason auth_failed, # observed twice). Write the file a real login would have left, # from the same token. expiresAt is only metadata here — the # setup-token itself carries the real expiry. run: | umask 077 mkdir -p "$HOME/.claude" EXPIRES=$(( ($(date +%s) + 250*24*3600) * 1000 )) printf '{"claudeAiOauth":{"accessToken":"%s","refreshToken":"","expiresAt":%s,"scopes":["user:inference"]}}' \ "$CLAUDE_CODE_OAUTH_TOKEN" "$EXPIRES" > "$HOME/.claude/.credentials.json" - name: run behavioral eval # --model pinned so a model rollout does not read as a plugin # regression; judge stays the default (haiku). Threshold 0.7 tolerates # one noisy judge vote without letting a broken skill pass. # # --trust-plugin answers the CLI's first-run trust prompt, which a # runner cannot answer interactively: since 2026-09-15 every run # exited 1 before the first case with "is not a trusted plugin # directory". The flag asserts trust in this plugin's own code and # eval suite — here the repository the workflow just checked out, # whose code this job already runs under the maintainer's # credential. It implies nothing else: not --scaffold, not # --allow-tools, and it does not turn mocks off. run: | claude plugin eval . --json results.json --threshold 0.7 \ --model claude-sonnet-5 --no-publish --trust-plugin - name: summarize if: always() run: | [ -f results.json ] || { echo "no results.json produced"; exit 0; } jq -r '"partial=\(.partial // false) reason=\(.partialReason // "none")"' results.json jq -r '.cases[].arms | to_entries[] | .key as $a | .value[] | select(.error != null) | " [\($a)] run error: \(.error)"' results.json jq -r '.cases[] | .name as $c | .arms | to_entries[] | "\($c) [\(.key)] score=" + ((([.value[].score] | add / length * 100 | round) / 100) | tostring)' results.json jq -r '.cases[].arms.with[].graders[] | " \(.name): passed=\(.passed) scored=\(.scored // true)"' results.json